Switches to our own version of the Url Detector

This commit is contained in:
Vitor Pamplona
2026-03-06 10:58:59 -05:00
parent 13f401df8c
commit f4d401bc56
32 changed files with 3794 additions and 156 deletions
@@ -54,7 +54,6 @@ Quartz exposes these as `api` (you get them transitively):
| `fr.acinq.secp256k1:secp256k1-kmp-*` | Schnorr signing |
| `com.github.anthonynsimon:rfc3986-normalizer` | Relay URL normalization |
| `com.fasterxml.jackson.module:jackson-module-kotlin` | Event JSON parsing |
| `com.linkedin.urls:url-detector` | URL extraction from content |
For Android, add to `build.gradle.kts`:
```kotlin
@@ -92,7 +92,6 @@ dependencies {
### JVM + Android (`jvmAndroid`)
- `com.github.anthonynsimon:rfc3986-normalizer` (URL normalization)
- `com.fasterxml.jackson.module:jackson-module-kotlin` (JSON)
- `com.linkedin.urls:url-detector` (URL extraction)
- `com.squareup.okhttp3:okhttp` (WebSocket)
- `ru.gildor.coroutines:kotlin-coroutines-okhttp`
- `nl.bommber:kchesslib` (NIP-64 chess, version pinned to 1.0.0)
@@ -24,6 +24,7 @@ import android.util.LruCache
import com.vitorpamplona.amethyst.commons.model.ImmutableListOfLists
import com.vitorpamplona.amethyst.commons.richtext.RichTextParser
import com.vitorpamplona.amethyst.commons.richtext.RichTextViewerState
import com.vitorpamplona.amethyst.commons.richtext.UrlParser
object CachedRichTextParser {
private val richTextCache = LruCache<Int, RichTextViewerState>(50)
@@ -77,7 +78,8 @@ object CachedUrlParser {
return if (cached != null) {
cached
} else {
val newUrls = RichTextParser().parseValidUrls(content).toList()
val urlSet = UrlParser().parseValidUrls(content)
val newUrls = urlSet.withScheme.toList() + urlSet.withoutScheme.map { "http://$it" }
parsedUrlsCache.put(key, newUrls)
newUrls
}
@@ -81,7 +81,6 @@ import com.vitorpamplona.amethyst.commons.richtext.PhoneSegment
import com.vitorpamplona.amethyst.commons.richtext.RegularTextSegment
import com.vitorpamplona.amethyst.commons.richtext.RelayUrlSegment
import com.vitorpamplona.amethyst.commons.richtext.RichTextViewerState
import com.vitorpamplona.amethyst.commons.richtext.SchemelessUrlSegment
import com.vitorpamplona.amethyst.commons.richtext.SecretEmoji
import com.vitorpamplona.amethyst.commons.richtext.Segment
import com.vitorpamplona.amethyst.commons.richtext.VideoSegment
@@ -219,8 +218,8 @@ fun RenderRegularPreview() {
// is HashIndexUserSegment -> TagLink(word, accountViewModel, nav)
// is HashIndexEventSegment -> TagLink(word, true, backgroundColorState, accountViewModel, nav)
is SchemelessUrlSegment -> {
NoProtocolUrlRenderer(word)
is LinkSegment -> {
ClickableUrl(word.segmentText, word.segmentText)
}
is RegularTextSegment -> {
@@ -257,7 +256,7 @@ fun RenderRegularPreview2() {
// is HashIndexUserSegment -> TagLink(word, accountViewModel, nav)
// is HashIndexEventSegment -> TagLink(word, true, backgroundColorState, accountViewModel, nav)
is SchemelessUrlSegment -> NoProtocolUrlRenderer(word)
is LinkSegment -> ClickableUrl(word.segmentText, word.segmentText)
is RegularTextSegment -> Text(word.segmentText)
@@ -307,7 +306,6 @@ fun RenderRegularPreview3() {
// is HashIndexUserSegment -> TagLink(word, accountViewModel, nav)
// is HashIndexEventSegment -> TagLink(word, true, backgroundColorState, accountViewModel, nav)
is SchemelessUrlSegment -> NoProtocolUrlRenderer(word)
is RegularTextSegment -> Text(word.segmentText)
@@ -485,8 +483,6 @@ private fun RenderWordWithoutPreview(
is HashIndexEventSegment -> TagLink(word, false, 0, backgroundColor, accountViewModel, nav)
is SchemelessUrlSegment -> NoProtocolUrlRenderer(word)
is RegularTextSegment -> Text(word.segmentText)
is RelayUrlSegment -> ClickableRelayUrl(word.segmentText, nav)
@@ -518,7 +514,6 @@ private fun RenderWordWithPreview(
is HashTagSegment -> HashTag(word, nav)
is HashIndexUserSegment -> TagLink(word, accountViewModel, nav)
is HashIndexEventSegment -> TagLink(word, true, quotesLeft, backgroundColor, accountViewModel, nav)
is SchemelessUrlSegment -> NoProtocolUrlRenderer(word)
is RegularTextSegment -> Text(word.segmentText)
is Base64Segment -> ZoomableContentView(word.segmentText, state, accountViewModel)
is RelayUrlSegment -> ClickableRelayUrl(word.segmentText, nav)
@@ -538,12 +533,6 @@ private fun ZoomableContentView(
}
}
@Composable
private fun NoProtocolUrlRenderer(segment: SchemelessUrlSegment) {
ClickableUrl(segment.url, "https://${segment.url}")
segment.extras?.let { it1 -> Text(it1) }
}
@Composable
fun RenderCustomEmoji(
word: String,
@@ -20,6 +20,7 @@
*/
package com.vitorpamplona.amethyst.ui.note.creators.previews
import androidx.compose.runtime.Stable
import androidx.compose.ui.text.input.TextFieldValue
import com.vitorpamplona.amethyst.service.CachedUrlParser
import kotlinx.coroutines.Dispatchers
@@ -30,6 +31,7 @@ import kotlinx.coroutines.flow.distinctUntilChanged
import kotlinx.coroutines.flow.flowOn
import kotlinx.coroutines.flow.map
@Stable
class PreviewState {
var source = MutableStateFlow(TextFieldValue(""))
@@ -30,9 +30,8 @@ import com.google.mlkit.nl.translate.TranslateLanguage
import com.google.mlkit.nl.translate.Translation
import com.google.mlkit.nl.translate.Translator
import com.google.mlkit.nl.translate.TranslatorOptions
import com.linkedin.urls.detection.UrlDetector
import com.linkedin.urls.detection.UrlDetectorOptions
import com.vitorpamplona.amethyst.service.checkNotInMainThread
import com.vitorpamplona.quartz.utils.urldetector.detection.UrlDetector
import kotlinx.coroutines.CancellationException
import java.util.concurrent.Executors
import java.util.regex.Pattern
@@ -185,8 +184,7 @@ object LanguageTranslatorService {
}
private fun urlDictionary(text: String): Map<String, String> {
val parser = UrlDetector(text, UrlDetectorOptions.Default)
val urlsInText = parser.detect()
val urlsInText = UrlDetector(text).detect()
var counter = 0
@@ -23,14 +23,14 @@ package com.vitorpamplona.amethyst.benchmark
import androidx.benchmark.junit4.BenchmarkRule
import androidx.benchmark.junit4.measureRepeated
import androidx.test.ext.junit.runners.AndroidJUnit4
import com.linkedin.urls.detection.UrlDetector
import com.linkedin.urls.detection.UrlDetectorOptions
import com.vitorpamplona.amethyst.commons.model.EmptyTagList
import com.vitorpamplona.amethyst.commons.model.ImmutableListOfLists
import com.vitorpamplona.amethyst.commons.richtext.HashTagSegment
import com.vitorpamplona.amethyst.commons.richtext.ImageSegment
import com.vitorpamplona.amethyst.commons.richtext.LinkSegment
import com.vitorpamplona.amethyst.commons.richtext.RichTextParser
import com.vitorpamplona.amethyst.commons.richtext.UrlParser
import com.vitorpamplona.quartz.utils.urldetector.detection.UrlDetector
import junit.framework.TestCase.assertNull
import org.junit.Assert.assertTrue
import org.junit.Rule
@@ -117,21 +117,21 @@ class RichTextParserBenchmark {
@Test
fun computeTestCase2UrlDetector() {
benchmarkRule.measureRepeated {
UrlDetector(testCase2, UrlDetectorOptions.Default).detect()
UrlDetector(testCase2).detect()
}
}
@Test
fun computeTestCase2UrlDetectorWJapanese() {
benchmarkRule.measureRepeated {
UrlDetector(testCaseJapanese, UrlDetectorOptions.Default).detect()
UrlDetector(testCaseJapanese).detect()
}
}
@Test
fun computeTestCase2ParseUrls() {
benchmarkRule.measureRepeated {
RichTextParser().parseValidUrls(testCase2)
UrlParser().parseValidUrls(testCase2)
}
}
-2
View File
@@ -86,8 +86,6 @@ kotlin {
create("jvmAndroid") {
dependsOn(commonMain.get())
dependencies {
// URL detection (JVM library, works on both)
implementation(libs.url.detector)
}
}
@@ -20,8 +20,6 @@
*/
package com.vitorpamplona.amethyst.commons.richtext
import com.linkedin.urls.detection.UrlDetector
import com.linkedin.urls.detection.UrlDetectorOptions
import com.vitorpamplona.amethyst.commons.emojicoder.EmojiCoder
import com.vitorpamplona.amethyst.commons.model.ImmutableListOfLists
import com.vitorpamplona.quartz.experimental.inlineMetadata.Nip54InlineMetadata
@@ -39,7 +37,6 @@ import kotlinx.collections.immutable.ImmutableList
import kotlinx.collections.immutable.persistentListOf
import kotlinx.collections.immutable.toImmutableList
import kotlinx.collections.immutable.toImmutableMap
import kotlinx.collections.immutable.toImmutableSet
import kotlinx.collections.immutable.toPersistentList
import java.net.MalformedURLException
import java.net.URISyntaxException
@@ -101,27 +98,43 @@ class RichTextParser {
}
}
fun parseValidUrls(content: String): LinkedHashSet<String> {
val urls = UrlDetector(content, UrlDetectorOptions.Default).detect()
fun fixMissingSpaces(
input: String,
urlList: Set<String>,
): String {
if (urlList.isEmpty()) return input
return urls.mapNotNullTo(LinkedHashSet(urls.size)) {
if (it.originalUrl.contains("@")) {
if (Patterns.EMAIL_ADDRESS.matches(it.originalUrl)) {
null
} else {
it.originalUrl
}
} else if (isNumber(it.originalUrl)) {
null // avoids urls that look like 123.22
} else if (it.originalUrl.contains("")) {
null // avoids Japanese characters as fake urls
} else {
if (HTTPRegex.matches(it.originalUrl)) {
it.originalUrl
} else {
null
}
// Escape and join words: (word1|word2)
val wordsPattern = urlList.sortedByDescending { it.length }.joinToString("|") { Regex.escape(it) }
// Regex breakdown:
// ([^ ])? -> Group 1: Optional character that is NOT a space or new line (Prefix)
// ($wordsPattern) -> Group 2: One of your target words
// ([^ ])? -> Group 3: Optional character that is NOT a space or new line (Suffix)
val regex = Regex("([^ \n])?($wordsPattern)([^ \n])?")
return regex.replace(input) { match ->
val prefix = match.groups[1]?.value ?: ""
val word = match.groups[2]?.value ?: ""
val suffix = match.groups[3]?.value ?: ""
val result = StringBuilder()
// Add prefix + space if the prefix exists
if (prefix.isNotEmpty()) {
result.append(prefix)
result.append(" ")
}
result.append(word)
// Add space + suffix if the suffix exists
if (suffix.isNotEmpty()) {
result.append(" ")
result.append(suffix)
}
result.toString()
}
}
@@ -131,30 +144,41 @@ class RichTextParser {
callbackUri: String?,
): RichTextViewerState {
val imetas = tags.lists.imetasByUrl()
val urlSet = parseValidUrls(content)
val urlSet = UrlParser().parseValidUrls(content)
val imagesForPager =
urlSet.mapNotNull { fullUrl -> createMediaContent(fullUrl, imetas, content, callbackUri) }.associateBy { it.url }
val mediaContents =
urlSet.withScheme.mapNotNull { fullUrl ->
createMediaContent(fullUrl, imetas, content, callbackUri)
} +
urlSet.withoutScheme.mapNotNull { fullUrl ->
createMediaContent(fullUrl, imetas, content, callbackUri)
}
val imageUrls = imagesForPager.filterValues { it is MediaUrlImage }.keys
val videoUrls = imagesForPager.filterValues { it is MediaUrlVideo }.keys
val mediaForPager = mediaContents.associateBy { it.url }
val imageUrls = mediaForPager.filterValues { it is MediaUrlImage }.keys
val videoUrls = mediaForPager.filterValues { it is MediaUrlVideo }.keys
val emojiMap = CustomEmoji.createEmojiMap(tags.lists)
val segments = findTextSegments(content, imageUrls, videoUrls, urlSet, emojiMap, tags)
val allUrls = urlSet.withScheme + urlSet.withoutScheme + urlSet.emails
val base64Images = segments.map { it.words.filterIsInstance<Base64Segment>() }.flatten()
val newContent = fixMissingSpaces(content, allUrls)
val imagesForPagerWithBase64 =
imagesForPager +
val segments = findTextSegments(newContent, imageUrls, videoUrls, urlSet, emojiMap, tags)
val base64Images = segments.flatMap { it.words.filterIsInstance<Base64Segment>() }
val mediaForPagerWithBase64 =
mediaForPager +
base64Images
.mapNotNull { createMediaContent(it.segmentText, emptyMap(), content, callbackUri) }
.associateBy { it.url }
return RichTextViewerState(
urlSet.toImmutableSet(),
imagesForPagerWithBase64.toImmutableMap(),
imagesForPagerWithBase64.values.toImmutableList(),
urlSet,
mediaForPagerWithBase64.toImmutableMap(),
mediaForPagerWithBase64.values.toImmutableList(),
emojiMap.toImmutableMap(),
segments,
tags,
@@ -165,7 +189,7 @@ class RichTextParser {
content: String,
images: Set<String>,
videos: Set<String>,
urls: Set<String>,
urls: Urls,
emojis: Map<String, String>,
tags: ImmutableListOfLists<String>,
): ImmutableList<ParagraphState> {
@@ -175,18 +199,14 @@ class RichTextParser {
lines.forEach { paragraph ->
val isRTL = isArabic(paragraph)
val wordList = paragraph.trimEnd().split(wordBoundaryRegex).filter { it.isNotEmpty() }
val wordList = paragraph.trimEnd().split(' ')
if (wordList.isEmpty()) {
paragraphSegments.add(ParagraphState(persistentListOf(RegularTextSegment("")), isRTL))
} else {
val segments = ArrayList<Segment>(wordList.size)
wordList.forEach { word ->
segments.add(wordIdentifier(word, images, videos, urls, emojis, tags))
}
paragraphSegments.add(ParagraphState(segments.toPersistentList(), isRTL))
val segments = ArrayList<Segment>(wordList.size)
wordList.forEach { word ->
segments.add(wordIdentifier(word, images, videos, urls, emojis, tags))
}
paragraphSegments.add(ParagraphState(segments.toPersistentList(), isRTL))
}
val segmentsWithGalleries = GalleryParser().processParagraphs(paragraphSegments)
@@ -204,8 +224,6 @@ class RichTextParser {
}.toImmutableList()
}
private fun isNumber(word: String) = numberPattern.matches(word)
private fun isPhoneNumberChar(c: Char): Boolean =
when (c) {
in '0'..'9' -> true
@@ -236,7 +254,7 @@ class RichTextParser {
word: String,
images: Set<String>,
videos: Set<String>,
urls: Set<String>,
urls: Urls,
emojis: Map<String, String>,
tags: ImmutableListOfLists<String>,
): Segment {
@@ -246,13 +264,29 @@ class RichTextParser {
if (Patterns.BASE64_IMAGE.matches(word)) return Base64Segment(word)
}
if (images.contains(word)) return ImageSegment(word)
if (images.contains(word)) {
return if (urls.withoutScheme.contains(word)) {
ImageSegment("https://$word")
} else {
ImageSegment(word)
}
}
if (videos.contains(word)) return VideoSegment(word)
if (videos.contains(word)) {
return if (urls.withoutScheme.contains(word)) {
VideoSegment("https://$word")
} else {
VideoSegment(word)
}
}
if (word.startsWith("ws://") || word.startsWith("wss://")) return RelayUrlSegment(word)
if (urls.contains(word)) return LinkSegment(word)
if (urls.withoutScheme.contains(word)) {
return LinkSegment("https://$word")
} else if (urls.withScheme.contains(word)) {
return LinkSegment(word)
}
if (CustomEmoji.fastMightContainEmoji(word, emojis) && emojis.any { word.contains(it.key) }) return EmojiSegment(word)
@@ -262,13 +296,11 @@ class RichTextParser {
if (word.startsWith("cashuA", true) || word.startsWith("cashuB", true)) return CashuSegment(word)
if (word.startsWith("#")) return parseHash(word, tags)
if (word.startsWith('#')) return parseHash(word, tags)
if (EmojiCoder.isCoded(word)) return SecretEmoji(word)
if (word.contains("@")) {
if (Patterns.EMAIL_ADDRESS.matches(word)) return EmailSegment(word)
}
if (urls.emails.contains(word)) return EmailSegment(word)
if (startsWithNIP19Scheme(word)) return BechSegment(word)
@@ -276,18 +308,6 @@ class RichTextParser {
if (Patterns.PHONE.matches(word)) return PhoneSegment(word)
}
val indexOfPeriod = word.indexOf(".")
if (indexOfPeriod > 0 && indexOfPeriod < word.length - 1) { // periods cannot be the last one
val schemelessMatcher = noProtocolUrlValidator.find(word)
if (schemelessMatcher != null) {
val url = schemelessMatcher.groups[1]?.value // url
val additionalChars = schemelessMatcher.groups[4]?.value?.ifEmpty { null } // additional chars
if (additionalUrlSchema.find(word) != null && url != null) {
return SchemelessUrlSegment(word, url, additionalChars)
}
}
}
return RegularTextSegment(word)
}
@@ -343,7 +363,7 @@ class RichTextParser {
val noProtocolUrlValidator =
Regex(
"(([a-zA-Z0-9_-]+\\.)*[a-zA-Z][a-zA-Z0-9_-]+[\\.\\:][a-zA-Z0-9_]+([\\/ \\?\\=\\&\\#\\.]?[a-zA-Z0-9_-]+)*\\/?)(.*)",
"(([a-zA-Z0-9_-]+@)?([a-zA-Z0-9_-]+\\.)*[a-zA-Z0-9_-]+[\\.\\:][a-zA-Z0-9_]+([\\/ \\?\\=\\&\\#\\.]?[a-zA-Z0-9_-]+)*\\/?)(.*)",
)
// Splits at spaces AND at ASCII/multibyte character boundaries
@@ -24,11 +24,10 @@ import androidx.compose.runtime.Immutable
import com.vitorpamplona.amethyst.commons.model.ImmutableListOfLists
import kotlinx.collections.immutable.ImmutableList
import kotlinx.collections.immutable.ImmutableMap
import kotlinx.collections.immutable.ImmutableSet
@Immutable
class RichTextViewerState(
val urlSet: ImmutableSet<String>,
val urlSet: Urls,
val imagesForPager: ImmutableMap<String, MediaUrlContent>,
val imageList: ImmutableList<MediaUrlContent>,
val customEmoji: ImmutableMap<String, String>,
@@ -139,13 +138,6 @@ class HashTagSegment(
val extras: String?,
) : Segment(segment)
@Immutable
class SchemelessUrlSegment(
segment: String,
val url: String,
val extras: String?,
) : Segment(segment)
@Immutable
class RelayUrlSegment(
segment: String,
@@ -0,0 +1,108 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.amethyst.commons.richtext
import com.vitorpamplona.amethyst.commons.richtext.RichTextParser.Companion.noProtocolUrlValidator
import com.vitorpamplona.quartz.utils.urldetector.Url
import com.vitorpamplona.quartz.utils.urldetector.detection.UrlDetector
class Urls(
val withScheme: Set<String> = emptySet(),
val withoutScheme: Set<String> = emptySet(),
val emails: Set<String> = emptySet(),
)
class UrlParser {
fun Char.isAsciiLetter(): Boolean = (this in 'a'..'z' || this in 'A'..'Z')
fun Url.isValidTopLevelDomain(): Boolean {
/*
According to the TLD Applicant Guidebook published June 2012, ICANN does not allow numbers in TLDs.
*/
val startOfTopDomain = host.lastIndexOf('.') + 1
return if (startOfTopDomain < host.length) {
val topLevelDomain = host.substring(startOfTopDomain)
topLevelDomain.isNotEmpty() && topLevelDomain[0].isAsciiLetter()
} else {
false
}
}
fun Url.wroteWithSchema(): Boolean = originalUrl.startsWith(scheme)
fun Url.isEmail(): Boolean = originalUrl.contains('@') && path == "/" && query.isEmpty() && fragment.isEmpty()
fun Char.isValidLastHostnameChar(): Boolean = (this in 'a'..'z' || this in 'A'..'Z' || this in '0'..'9')
fun Url.isValidLastHostnameChar(): Boolean = host[host.length - 1].isValidLastHostnameChar()
fun Url.endsWithHost(): Boolean = originalUrl.endsWith(host)
val notAHostNameChar = "[^a-zA-Z0-9.-]".toRegex()
fun parseValidUrls(content: String): Urls {
val urls = UrlDetector(content).detect()
val completeUrls = mutableSetOf<String>()
val urlsWithoutScheme = mutableSetOf<String>()
val emails = mutableSetOf<String>()
println("AABBBCC parseValidUrls ${urls.size}")
urls.forEach {
println("AABBBCC Testing ${it.originalUrl}")
if (it.isValidTopLevelDomain()) {
if (it.wroteWithSchema()) {
if (it.isValidLastHostnameChar()) {
completeUrls.add(it.originalUrl)
} else if (it.endsWithHost()) {
val match = notAHostNameChar.find(it.host)
if (match != null) {
completeUrls.add(it.originalUrl.substring(0, (it.originalUrl.length - it.host.length) + match.range.first))
} else {
completeUrls.add(it.originalUrl)
}
}
} else {
// emails are understood as urls from the detector.
if (it.isEmail()) {
Patterns.EMAIL_ADDRESS.findAll(it.originalUrl).forEach {
emails.add(it.value)
}
} else {
noProtocolUrlValidator.findAll(it.originalUrl).forEach { components ->
val url = components.groups[1]?.value
if (url != null) {
urlsWithoutScheme.add(url)
}
}
}
}
}
}
return Urls(
withScheme = completeUrls,
withoutScheme = urlsWithoutScheme,
emails = emails,
)
}
}
@@ -1,3 +1,23 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.amethyst.commons.richtext
import kotlin.test.Test
@@ -55,7 +55,55 @@ class RichTextParserMultibyteTest {
// user@example.com should not be in urlSet
assertTrue(
"user@example.com should not be in urlSet",
!state.urlSet.contains("user@example.com"),
!state.urlSet.withScheme.contains("user@example.com") && !state.urlSet.withoutScheme.contains("user@example.com"),
)
}
@Test
fun testHttpWithoutSpaces() {
// Multibyte characters around an email address should not produce URL/Link segments
val text =
"Vitor, vocêhttp://test.com? \uD83E\uDD7A"
val state =
RichTextParser()
.parseText(text, EmptyTagList, null)
assertEquals(
"Vitor, você http://test.com? \uD83E\uDD7A",
state.paragraphs.joinToString("\n") { it.words.joinToString(" ") { it.segmentText } },
)
}
@Test
fun testHttpWithoutSpacesJapan() {
// Multibyte characters around an email address should not produce URL/Link segments
val text =
"Vitor, vocêhttp://test.comほげほげ"
val state =
RichTextParser()
.parseText(text, EmptyTagList, null)
assertEquals(
"Vitor, você http://test.com ほげほげ",
state.paragraphs.joinToString("\n") { it.words.joinToString(" ") { it.segmentText } },
)
}
@Test
fun testHttpWithoutSpacesJapan2() {
// Multibyte characters around an email address should not produce URL/Link segments
val text =
"Vitor, vocêhttp://test.com。ほげほげ"
val state =
RichTextParser()
.parseText(text, EmptyTagList, null)
assertEquals(
"Vitor, você http://test.com 。ほげほげ",
state.paragraphs.joinToString("\n") { it.words.joinToString(" ") { it.segmentText } },
)
}
@@ -65,11 +113,29 @@ class RichTextParserMultibyteTest {
val text =
"Vitor, você tem como colocar alguma forma de aviso se o link vai carregar uma imagem ou um vídeo? \uD83E\uDD7A"
val regex = Regex("(?<=[\\u0000-\\u00FF])(?=[\\u0100-\\uFFFF])|(?<=[\\u0100-\\uFFFF])(?=[\\u0000-\\u00FF])| +")
val state =
RichTextParser()
.parseText(text, EmptyTagList, null)
assertEquals(
"Vitor,-você-tem-como-colocar-alguma-forma-de-aviso-se-o-link-vai-carregar-uma-imagem-ou-um-vídeo?-\uD83E\uDD7A",
text.split(regex).joinToString("-"),
"Vitor, você tem como colocar alguma forma de aviso se o link vai carregar uma imagem ou um vídeo? \uD83E\uDD7A",
state.paragraphs.joinToString("\n") { it.words.joinToString(" ") { it.segmentText } },
)
}
@Test
fun testFullTextWithMultibyteAndQuotes() {
// Multibyte characters around an email address should not produce URL/Link segments
val text =
"Ive been thinking lately about how I believe there will more than likely be models unattainable by most. Think Bloomberg Terminal. Where their cost of tokens is too high for the average lay person, but their level of “cognition” is unmatched by anything else. Im sure there will even be many closed models that are invite only. Crazy times ahead."
val state =
RichTextParser()
.parseText(text, EmptyTagList, null)
assertEquals(
"Ive been thinking lately about how I believe there will more than likely be models unattainable by most. Think Bloomberg Terminal. Where their cost of tokens is too high for the average lay person, but their level of “cognition” is unmatched by anything else. Im sure there will even be many closed models that are invite only. Crazy times ahead.",
state.paragraphs.joinToString("\n") { it.words.joinToString(" ") { it.segmentText } },
)
}
@@ -91,9 +157,9 @@ class RichTextParserMultibyteTest {
val state = RichTextParser().parseText(text, EmptyTagList, null)
val allSegments = state.paragraphs.flatMap { it.words }
val urlSegments = allSegments.filterIsInstance<SchemelessUrlSegment>()
val urlSegments = allSegments.filterIsInstance<LinkSegment>()
assertTrue("Should have SchemelessUrlSegment", urlSegments.isNotEmpty())
assertTrue("URL should be example.com", urlSegments.any { it.url == "example.com" })
assertTrue("URL should be example.com", urlSegments.any { it.segmentText == "https://example.com" })
val textSegments = allSegments.filterIsInstance<RegularTextSegment>()
assertTrue("Should have prefix ああ", textSegments.any { it.segmentText == "ああ" })
@@ -106,9 +172,9 @@ class RichTextParserMultibyteTest {
val state = RichTextParser().parseText(text, EmptyTagList, null)
val allSegments = state.paragraphs.flatMap { it.words }
val urlSegments = allSegments.filterIsInstance<SchemelessUrlSegment>()
val urlSegments = allSegments.filterIsInstance<LinkSegment>()
assertTrue("Should have SchemelessUrlSegment", urlSegments.isNotEmpty())
assertTrue("URL should be example.com", urlSegments.any { it.url == "example.com" })
assertTrue("URL should be example.com", urlSegments.any { it.segmentText == "https://example.com" })
val textSegments = allSegments.filterIsInstance<RegularTextSegment>()
assertTrue("Should have suffix ああ", textSegments.any { it.segmentText == "ああ" })
File diff suppressed because one or more lines are too long
@@ -0,0 +1,290 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.amethyst.commons.richtext
import kotlin.test.Ignore
import kotlin.test.Test
import kotlin.test.assertEquals
class UrlParserTest {
val parser = UrlParser()
fun test(
text: String,
expected: Urls,
) {
val urlSet = parser.parseValidUrls(text)
assertEquals(expected.withScheme, urlSet.withScheme)
assertEquals(expected.withoutScheme, urlSet.withoutScheme)
assertEquals(expected.emails, urlSet.emails)
}
@Test
fun testSimpleText() =
test(
"test. com",
Urls(),
)
@Test
fun testBasicUrl() =
test(
"http://test.com",
Urls(withScheme = setOf("http://test.com")),
)
@Test
fun testNoSchemaUrl() =
test(
"test.com",
Urls(withoutScheme = setOf("test.com")),
)
@Test
fun testNoSchemaUrlPrefixMultibyte() =
test(
"ほtest.com",
Urls(withoutScheme = setOf("test.com")),
)
@Test
fun testNoSchemaUrlSuffixMultibyte() =
test(
"test.comほ",
Urls(withoutScheme = setOf("test.com")),
)
@Test
fun testNoSchemaUrlWithParams() =
test(
"test.com/some/me/hey?param=value#some=value",
Urls(withoutScheme = setOf("test.com/some/me/hey?param=value#some=value")),
)
@Test
fun testNoSchemaUrlWithParamsWithOtherWords() =
test(
"Hi there, check my website test.com/some/me/hey?param=value#some=value .",
Urls(withoutScheme = setOf("test.com/some/me/hey?param=value#some=value")),
)
@Test
fun testBasicUrlWithoutSpaceBefore() =
test(
"ahttp://test.com",
Urls(withScheme = setOf("http://test.com")),
)
@Test
fun testBasicUrlWithoutSpaceBeforeMultiByte() =
test(
"ほhttp://test.com",
Urls(withScheme = setOf("http://test.com")),
)
@Test
fun testBasicUrlWithoutSpaceAfter() =
test(
"http://test.comほ",
Urls(withScheme = setOf("http://test.com")),
)
@Test
fun testBasicUrlWithMultibytePath() =
test(
"http://test.com/ほ",
Urls(withScheme = setOf("http://test.com/ほ")),
)
@Test
fun testBasicUrls() =
test(
"http://test.com http://test2.com",
Urls(withScheme = setOf("http://test.com", "http://test2.com")),
)
@Test
fun testEmail() =
test(
"vitor@vitorpamplona.com",
Urls(emails = setOf("vitor@vitorpamplona.com")),
)
@Test
fun testEmailWithMultibytePrefix() =
test(
"ほvitor@vitorpamplona.com",
Urls(emails = setOf("vitor@vitorpamplona.com")),
)
@Test
fun testEmailWithMultibyteSuffix() =
test(
"vitor@vitorpamplona.comほ",
Urls(emails = setOf("vitor@vitorpamplona.com")),
)
@Test
fun testEmailWithMultibyteBoth() =
test(
"ほvitor@vitorpamplona.comほ",
Urls(emails = setOf("vitor@vitorpamplona.com")),
)
@Test
fun testUrlWithUserAndMultibyteSuffix() =
test(
"http://vitor@vitorpamplona.comほ",
Urls(withScheme = setOf("http://vitor@vitorpamplona.com")),
)
@Test
fun testUrlWithUserAndMultibytePrefix() =
test(
"ほhttp://vitor@vitorpamplona.com",
Urls(withScheme = setOf("http://vitor@vitorpamplona.com")),
)
@Test
fun testNostrUrls() =
test(
"nostr:npub1aabbcc",
Urls(),
)
@Test
fun testUrlsWithUsernameAndPath() =
test(
"miceliomad@miceliomad.github.io/nostr/",
Urls(withoutScheme = setOf("miceliomad@miceliomad.github.io/nostr/")),
)
@Test
fun testUrlsWithQuery() =
test(
" universe.nostrich.land?lang=zh ",
Urls(withoutScheme = setOf("universe.nostrich.land?lang=zh")),
)
@Test
fun testUrlsWithSchemaPathAndQuery() =
test(
"https://miceliomad.github.io/nostr/test?me=you",
Urls(withScheme = setOf("https://miceliomad.github.io/nostr/test?me=you")),
)
@Test
fun testUrlsWithPathAndQuery() =
test(
"miceliomad.github.io/nostr/test?me=you",
Urls(withoutScheme = setOf("miceliomad.github.io/nostr/test?me=you")),
)
@Test
fun testAvifFileNameComplete() =
test(
"https://bae.st/media/66b08dde784287ed8f92c455bc62076a04671ccb44097550626a532185a5d3ed.avif?name=81ca16-b665-4f57-80cb-11a58461fb61.avif",
Urls(withScheme = setOf("https://bae.st/media/66b08dde784287ed8f92c455bc62076a04671ccb44097550626a532185a5d3ed.avif?name=81ca16-b665-4f57-80cb-11a58461fb61.avif")),
)
@Test
fun testAvifFileName() =
test(
"81ca16-b665-4f57-80cb-11a58461fb61.avif",
Urls(withoutScheme = setOf("81ca16-b665-4f57-80cb-11a58461fb61.avif")),
)
@Test
fun testMultiLine() =
test(
"""
22.8K (3.2%) nos.lol
22.7K (3.1%) universe.nostrich.land?lang=zh
22.5K (3.1%) universe.nostrich.land?lang=en
""".trimIndent(),
Urls(withoutScheme = setOf("nos.lol", "universe.nostrich.land?lang=zh", "universe.nostrich.land?lang=en")),
)
@Test
fun testEmailWithDashes() =
test(
"freeverification@Nostr-Check.com",
Urls(emails = setOf("freeverification@Nostr-Check.com")),
)
@Test
fun testEmailWithManyDashes() =
test(
"free-veri-fica-tion@No-str-Ch-eck.com",
Urls(emails = setOf("free-veri-fica-tion@No-str-Ch-eck.com")),
)
@Test
fun testEmailWithUnderscore() =
test(
"vi_t_or@vitorpamplona.com",
Urls(emails = setOf("vi_t_or@vitorpamplona.com")),
)
@Test
fun testEmailsWithPeriod() =
test(
"john.smith@gmail.com",
Urls(emails = setOf("john.smith@gmail.com")),
)
@Test
fun testStrangeError() =
test(
"neomobius_at_mstdn.jp@mostr.pub",
Urls(emails = setOf("neomobius_at_mstdn.jp@mostr.pub")),
)
@Test
@Ignore("We need to make this work")
fun testRelayUrl() =
test(
"wss://test.com",
Urls(withScheme = setOf("wss://test.com")),
)
@Test
@Ignore("We need to make this work")
fun testBech12() =
test(
"nostr:npub1aabbcc",
Urls(withScheme = setOf("wss://test.com")),
)
@Test
fun testJapaneseUrls() =
test(
"我进入你的主页很卡顿,也许是你的关注人数或者其他数据太多了,其他人主页没有这么卡顿。来自amethyst客户端",
Urls(withScheme = emptySet()),
)
@Test
fun testHour() =
test(
"10.00hr,",
Urls(withScheme = emptySet()),
)
}
-2
View File
@@ -54,7 +54,6 @@ tarsosdsp = "2.5"
torAndroid = "0.4.9.5"
translate = "17.0.3"
unifiedpush = "3.0.10"
urlDetector = "0.1.23"
vico-charts = "2.4.3"
zelory = "3.0.1"
zoomable = "2.11.1"
@@ -154,7 +153,6 @@ secp256k1-kmp-jni-jvm = { group = "fr.acinq.secp256k1", name = "secp256k1-kmp-jn
tarsosdsp = { group = "be.tarsos.dsp", name = "core", version.ref = "tarsosdsp" }
tor-android = { module = "info.guardianproject:tor-android", version.ref = "torAndroid" }
unifiedpush = { group = "com.github.UnifiedPush", name = "android-connector", version.ref = "unifiedpush" }
url-detector = { group = "io.github.url-detector", name = "url-detector", version.ref = "urlDetector" }
vico-charts-compose = { group = "com.patrykandpatrick.vico", name = "compose", version.ref = "vico-charts" }
vico-charts-core = { group = "com.patrykandpatrick.vico", name = "core", version.ref = "vico-charts" }
vico-charts-m3 = { group = "com.patrykandpatrick.vico", name = "compose-m3", version.ref = "vico-charts" }
+10
View File
@@ -0,0 +1,10 @@
# Default ignored files
/shelf/
/workspace.xml
# Ignored default folder with query files
/queries/
# Datasource local storage ignored files
/dataSources/
/dataSources.local.xml
# Editor-based HTTP Client requests
/httpRequests/
-3
View File
@@ -161,9 +161,6 @@ kotlin {
// Performant Parser of JSONs into Events
api(libs.jackson.module.kotlin)
// Parses URLs from Text:
api(libs.url.detector)
// Websockets API
implementation(libs.okhttp)
implementation(libs.okhttpCoroutines)
@@ -0,0 +1,248 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector
/**
* Creating own Uri class since java.net.Uri would throw parsing exceptions
* for URL's considered ok by browsers.
*
* Also to avoid further conflict, this does stuff that the normal Uri object doesn't do:
* - Converts http://google.com/a/b/.//./../c to http://google.com/a/c
* - Decodes repeatedly so that http://host/%2525252525252525 becomes http://host/%25 while normal decoders
* would make it http://host/%25252525252525 (one less 25)
* - Removes tabs and new lines: http://www.google.com/foo\tbar\rbaz\n2 becomes "http://www.google.com/foobarbaz2"
* - Converts IP addresses: http://3279880203/blah becomes http://195.127.0.11/blah
* - Strips fragments (anything after #)
*
*/
class Url(
val urlMarker: UrlMarker,
) {
private var _scheme: String? = null
private var _username: String? = null
private var _password: String? = null
private var rawHost: String? = null
private var _port = 0
private var rawPath: String? = null
private var _query: String? = null
private var _fragment: String? = null
val originalUrl: String = urlMarker.originalUrl
override fun toString(): String = this.fullUrl
/**
* Note that this includes the fragment
* @return Formats the url to: [scheme]://[username]:[password]@[host]:[port]/[path]?[query]#[fragment]
*/
val fullUrl: String
get() = this.fullUrlWithoutFragment + this.fragment
/**
*
* @return Formats the url to: [scheme]://[username]:[password]@[host]:[port]/[path]?[query]
*/
val fullUrlWithoutFragment: String
get() {
val url = StringBuilder()
if (this.scheme.isNotEmpty()) {
url.append(this.scheme)
url.append(":")
}
url.append("//")
if (this.username.isNotEmpty()) {
url.append(this.username)
if (this.password.isNotEmpty()) {
url.append(":")
url.append(this.password)
}
url.append("@")
}
url.append(this.host)
if (this.port > 0 && this.port != SCHEME_PORT_MAP[this.scheme]) {
url.append(":")
url.append(this.port)
}
url.append(this.path)
url.append(this.query)
return url.toString()
}
val scheme: String
get() {
if (_scheme == null) {
if (exists(UrlPart.SCHEME)) {
_scheme = getPart(UrlPart.SCHEME)
val index = _scheme!!.indexOf(":")
if (index != -1) {
_scheme = _scheme!!.substring(0, index)
}
} else if (!originalUrl.startsWith("//")) {
_scheme =
DEFAULT_SCHEME
}
}
return _scheme ?: ""
}
val username: String
get() {
if (_username == null) {
populateUsernamePassword()
}
return _username ?: ""
}
val password: String
get() {
if (_password == null) {
populateUsernamePassword()
}
return _password ?: ""
}
val host: String
get() {
if (this.rawHost == null) {
this.rawHost = getPart(UrlPart.HOST)
if (exists(UrlPart.PORT)) {
this.rawHost =
rawHost?.let {
it.substring(0, it.length - 1)
}
}
}
return this.rawHost!!
}
/**
* port = 0 means it hasn't been set yet. port = -1 means there is no port
*/
val port: Int
get() {
if (_port == 0) {
val portString =
getPart(UrlPart.PORT)
if (!portString.isNullOrEmpty()) {
_port = portString.toIntOrNull() ?: -1
} else {
_port = SCHEME_PORT_MAP[this.scheme] ?: -1
}
}
return _port
}
val path: String?
get() {
if (this.rawPath == null) {
this.rawPath =
if (exists(UrlPart.PATH)) {
getPart(
UrlPart.PATH,
)
} else {
"/"
}
}
return this.rawPath
}
val query: String
get() {
if (_query == null) {
_query = getPart(UrlPart.QUERY)
}
return _query ?: ""
}
val fragment: String
get() {
if (_fragment == null) {
_fragment = getPart(UrlPart.FRAGMENT)
}
return _fragment ?: ""
}
private fun populateUsernamePassword() {
val usernamePassword = getPart(UrlPart.USERNAME_PASSWORD)
if (usernamePassword != null) {
val usernamePasswordParts: List<String> =
usernamePassword.substring(0, usernamePassword.length - 1).split(":")
if (usernamePasswordParts.size == 1) {
_username = usernamePasswordParts[0]
} else if (usernamePasswordParts.size == 2) {
_username = usernamePasswordParts[0]
_password = usernamePasswordParts[1]
}
}
}
/**
* @param urlPart The url part we are checking for existence
* @return Returns true if the part exists.
*/
private fun exists(urlPart: UrlPart?): Boolean = urlPart != null && urlMarker.indexOf(urlPart) >= 0
/**
* For example, in http://yahoo.com/lala/, nextExistingPart(UrlPart.HOST) would return UrlPart.PATH
* @param urlPart The current url part
* @return Returns the next part; if there is no existing next part, it returns null
*/
private fun nextExistingPart(urlPart: UrlPart): UrlPart? {
val nextPart = urlPart.nextPart
if (exists(nextPart)) {
return nextPart
} else if (nextPart == null) {
return null
} else {
return nextExistingPart(nextPart)
}
}
/**
* @param part The part that we want. Ex: host, path
*/
private fun getPart(part: UrlPart): String? {
if (!exists(part)) {
return null
}
val nextPart = nextExistingPart(part)
return if (nextPart == null) {
originalUrl.substring(urlMarker.indexOf(part))
} else {
originalUrl.substring(urlMarker.indexOf(part), urlMarker.indexOf(nextPart))
}
}
companion object {
private const val DEFAULT_SCHEME = "https"
private val SCHEME_PORT_MAP: Map<String, Int> =
mapOf(
"http" to 80,
"https" to 443,
"ftp" to 21,
)
}
}
@@ -0,0 +1,84 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector
class UrlMarker {
private var schemeIndex = -1
private var usernamePasswordIndex = -1
private var hostIndex = -1
private var portIndex = -1
private var pathIndex = -1
private var queryIndex = -1
private var fragmentIndex = -1
var originalUrl: String = ""
fun createUrl(): Url = Url(this)
fun setIndex(
urlPart: UrlPart,
index: Int,
) {
when (urlPart) {
UrlPart.SCHEME -> schemeIndex = index
UrlPart.USERNAME_PASSWORD -> usernamePasswordIndex = index
UrlPart.HOST -> hostIndex = index
UrlPart.PORT -> portIndex = index
UrlPart.PATH -> pathIndex = index
UrlPart.QUERY -> queryIndex = index
UrlPart.FRAGMENT -> fragmentIndex = index
}
}
/**
* @param urlPart The part you want the index of
* @return Returns the index of the part
*/
fun indexOf(urlPart: UrlPart): Int =
when (urlPart) {
UrlPart.SCHEME -> schemeIndex
UrlPart.USERNAME_PASSWORD -> usernamePasswordIndex
UrlPart.HOST -> hostIndex
UrlPart.PORT -> portIndex
UrlPart.PATH -> pathIndex
UrlPart.QUERY -> queryIndex
UrlPart.FRAGMENT -> fragmentIndex
}
fun unsetIndex(urlPart: UrlPart) {
setIndex(urlPart, -1)
}
/**
* This is used in TestUrlMarker to set indices more easily.
* @param indices array of indices of size 7
*/
fun setIndices(indices: IntArray): UrlMarker {
require(indices.size == 7) { "Malformed index array." }
setIndex(UrlPart.SCHEME, indices[0])
setIndex(UrlPart.USERNAME_PASSWORD, indices[1])
setIndex(UrlPart.HOST, indices[2])
setIndex(UrlPart.PORT, indices[3])
setIndex(UrlPart.PATH, indices[4])
setIndex(UrlPart.QUERY, indices[5])
setIndex(UrlPart.FRAGMENT, indices[6])
return this
}
}
@@ -0,0 +1,36 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector
enum class UrlPart(
nextPart: UrlPart?,
) {
FRAGMENT(null),
QUERY(FRAGMENT),
PATH(QUERY),
PORT(PATH),
HOST(PORT),
USERNAME_PASSWORD(HOST),
SCHEME(USERNAME_PASSWORD),
;
val nextPart: UrlPart? = nextPart
}
@@ -0,0 +1,86 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector.detection
object CharUtils {
/**
* Checks if character is a valid hex character.
*/
fun isHex(a: Char): Boolean = (a in '0'..'9') || (a in 'a'..'f') || (a in 'A'..'F')
/**
* Checks if character is a valid alphabetic character.
*/
fun isAlpha(a: Char): Boolean = ((a in 'a'..'z') || (a in 'A'..'Z'))
/**
* Checks if character is a valid numeric character.
*/
fun isNumeric(a: Char): Boolean = a in '0'..'9'
/**
* Checks if character is a valid alphanumeric character.
*/
fun isAlphaNumeric(a: Char): Boolean = isAlpha(a) || isNumeric(a)
/**
* Checks if character is a valid unreserved character. This is defined by the RFC 3986 ABNF
*/
fun isUnreserved(a: Char): Boolean = isAlphaNumeric(a) || a == '-' || a == '.' || a == '_' || a == '~'
/**
* Checks if character is a dot. Heres the doc:
* http://docs.oracle.com/javase/6/docs/api/java/net/IDN.html#toASCII%28java.lang.String,%20int%29
*/
fun isDot(a: Char): Boolean = (a == '.' || a == '\u3002' || a == '\uFF0E' || a == '\uFF61')
fun isWhiteSpace(a: Char): Boolean = (a == '\n' || a == '\t' || a == '\r' || a == ' ')
/**
* Splits a string without the use of a regex, which could split either by isDot() or %2e
* @param input the input string that will be split by dot
* @return an array of strings that is a partition of the original string split by dot
*/
fun splitByDot(input: String): List<String> {
val splitList = ArrayList<String>()
val section = StringBuilder()
if (input.isEmpty()) {
return listOf("")
}
val reader = InputTextReader(input)
while (!reader.eof()) {
val curr: Char = reader.read()
if (isDot(curr)) {
splitList.add(section.toString())
section.setLength(0)
} else if (curr == '%' && (reader.peekEquals("2e") || reader.peekEquals("2E"))) {
reader.read()
reader.read() // advance past the 2e
splitList.add(section.toString())
section.setLength(0)
} else {
section.append(curr)
}
}
splitList.add(section.toString())
return splitList
}
}
@@ -0,0 +1,760 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector.detection
import com.vitorpamplona.quartz.utils.urldetector.detection.CharUtils.isAlpha
import com.vitorpamplona.quartz.utils.urldetector.detection.CharUtils.isAlphaNumeric
import com.vitorpamplona.quartz.utils.urldetector.detection.CharUtils.isDot
import com.vitorpamplona.quartz.utils.urldetector.detection.CharUtils.isHex
import com.vitorpamplona.quartz.utils.urldetector.detection.CharUtils.isNumeric
import com.vitorpamplona.quartz.utils.urldetector.detection.CharUtils.isUnreserved
import com.vitorpamplona.quartz.utils.urldetector.detection.CharUtils.splitByDot
import kotlin.math.max
import kotlin.math.min
/**
* The domain name reader reads input from a InputTextReader and validates if the content being read is a valid domain name.
* After a domain name is read, the returning status is what to do next. If the domain is valid but a specific character is found,
* the next state will be to read another part for the rest of the url. For example, if a "?" is found at the end and the
* domain is valid, the return state will be to read a query string.
*/
class DomainNameReader(
val reader: InputTextReader,
/**
* The currently written string buffer.
*/
val buffer: StringBuilder,
/**
* The domain name started with a partial domain name found. This is the original string of the domain name only.
*/
val current: String?,
) {
/**
* This is the final return state of reading a domain name.
*/
enum class ReaderNextState {
/**
* Trying to read the domain name caused it to be invalid.
*/
InvalidDomainName,
/**
* The domain name is found to be valid.
*/
ValidDomainName,
/**
* Finished reading, next step should be to read the fragment.
*/
ReadFragment,
/**
* Finished reading, next step should be to read the path.
*/
ReadPath,
/**
* Finished reading, next step should be to read the port.
*/
ReadPort,
/**
* Finished reading, next step should be to read the query string.
*/
ReadQueryString,
/**
* This was actually not a domain at all.
*/
ReadUserPass,
}
/**
* The interface that gets called for each character that's non-matching (to a valid domain name character) in to count
* the matching quotes and parenthesis correctly.
*/
interface CharacterHandler {
fun addCharacter(character: Char)
}
/**
* Keeps track the number of dots that were found in the domain name.
*/
private var dots = 0
/**
* Keeps track of the number of characters since the last "."
*/
private var currentLabelLength = 0
/**
* Keeps track of the number of characters in the top level domain.
*/
private var topLevelLength = 0
/**
* Keeps track where the domain name started. This is non zero if the buffer starts with
* http://username:password@...
*/
private var startDomainName = 0
/**
* Keeps track if the entire domain name is numeric.
*/
private var numeric = false
/**
* Keeps track if we are seeing an ipv6 type address.
*/
private var seenBracket = false
/**
* Keeps track if we have seen a full bracket set "[....]"; used for ipv6 type address.
*/
private var seenCompleteBracketSet = false
/**
* Keeps track if we have a zone index in the ipv6 address.
*/
private var zoneIndex = false
fun String.isDotPercent() = this == "%2e" || this == "%2E"
/**
* Reads and parses the current string to make sure the domain name started where it was supposed to,
* and the current domain name is correct.
* @return The next state to use after reading the current.
*/
private fun readCurrent(): ReaderNextState {
if (current != null) {
// Handles the case where the string is ".hello"
if (current.length == 1 && isDot(current[0])) {
return ReaderNextState.InvalidDomainName
} else if (current.length == 3 && current.isDotPercent()) {
return ReaderNextState.InvalidDomainName
}
// The location where the domain name started.
startDomainName = buffer.length - current.length
// flag that the domain is currently all numbers and/or dots.
numeric = true
// If an invalid char is found, we can just restart the domain from there.
var newStart = 0
val currArray = current.toCharArray()
val length = currArray.size
// hex special case
var isAllHexSoFar =
length > 2 && (currArray[0] == '0' && (currArray[1] == 'x' || currArray[1] == 'X'))
var index = if (isAllHexSoFar) 2 else 0
var done = false
while (index < length && !done) {
// get the current character and update length counts.
val curr = currArray[index]
currentLabelLength++
topLevelLength = currentLabelLength
// Is the length of the last part > 64 (plus one since we just incremented)
if (currentLabelLength > MAX_LABEL_LENGTH) {
return ReaderNextState.InvalidDomainName
} else if (isDot(curr)) {
// found a dot. Increment dot count, and reset last length
dots++
currentLabelLength = 0
} else if (curr == '[') {
seenBracket = true
numeric = false
} else if (curr == '%' && index + 2 < length && isHex(currArray[index + 1]) &&
isHex(currArray[index + 2])
) {
// handle url encoded dot
if (currArray[index + 1] == '2' && currArray[index + 2] == 'e') {
dots++
currentLabelLength = 0
} else {
numeric = false
}
index += 2
} else if (isAllHexSoFar) {
// if it's a valid character in the domain that is not numeric
if (!isHex(curr)) {
numeric = false
isAllHexSoFar = false
index-- // backtrack to rerun last character knowing it isn't hex.
}
} else if (isAlpha(curr) || curr == '-' || curr.code >= INTERNATIONAL_CHAR_START) {
numeric = false
} else if (!isNumeric(curr)) {
// if its not _numeric and not alphabetical, then restart searching for a domain from this point.
newStart = index + 1
currentLabelLength = 0
topLevelLength = 0
numeric = true
dots = 0
done = true
}
index++
}
// An invalid character for the domain was found somewhere in the current buffer.
// cut the first part of the domain out. For example:
// http://asdf%asdf.google.com <- asdf.google.com is still valid, so restart from the %
if (newStart > 0) {
// make sure the location is not at the end. Otherwise the thing is just invalid.
if (newStart < current.length) {
buffer.replaceRange(0, buffer.length, current.substring(newStart))
// cut out the previous part, so now the domain name has to be from here.
startDomainName = 0
}
// now after cutting if the buffer is just "." newStart > current (last character in current is invalid)
if (newStart >= current.length || buffer.toString() == ".") {
return ReaderNextState.InvalidDomainName
}
}
} else {
startDomainName = buffer.length
}
// all else is good, return OK
return ReaderNextState.ValidDomainName
}
/**
* Reads the Dns and returns the next state the state machine should take in throwing this out, or continue processing
* if this is a valid domain name.
* @return The next state to take.
*/
fun readDomainName(): ReaderNextState? {
// Read the current, and if its bad, just return.
if (readCurrent() == ReaderNextState.InvalidDomainName) {
return ReaderNextState.InvalidDomainName
}
// while not done and not end of string keep reading.
var done = false
// If this is the first domain part, check if it's ip address in is hexa
// similar to what is done on 'readCurrent' method
val isAllHexSoFar =
(current == null || current == "") &&
reader.canReadChars(3) &&
(reader.peekEquals("0x") || reader.peekEquals("0X"))
if (isAllHexSoFar) {
// Append hexa radix symbol characters (0x)
buffer.append(reader.read())
buffer.append(reader.read())
currentLabelLength += 2
topLevelLength = currentLabelLength
}
while (!done && !reader.eof()) {
val curr: Char = reader.read()
if (curr == '/') {
// continue by reading the path
return checkDomainNameValid(ReaderNextState.ReadPath, curr)
} else if (curr == ':' && (!seenBracket || seenCompleteBracketSet)) {
// Don't check for a port if it's in the middle of an ipv6 address
// continue by reading the port.
return checkDomainNameValid(ReaderNextState.ReadPort, curr)
} else if (curr == '?') {
// continue by reading the query string
return checkDomainNameValid(ReaderNextState.ReadQueryString, curr)
} else if (curr == '#') {
// continue by reading the fragment
return checkDomainNameValid(ReaderNextState.ReadFragment, curr)
} else if (curr == '@') {
// this may not have been a domain after all, but rather a username/password instead
reader.goBack()
return ReaderNextState.ReadUserPass
} else if (isDot(curr) || (curr == '%' && (reader.peekEquals("2e") || reader.peekEquals("2E")))) {
// if the current character is a dot or a urlEncodedDot
// handles the case: hello..
if (currentLabelLength < 1) {
done = true
} else {
// append the "." to the domain name
buffer.append(curr)
// if it was not a normal dot, then it is url encoded
// read the next two chars, which are the hex representation
if (!isDot(curr)) {
buffer.append(reader.read())
buffer.append(reader.read())
}
// increment the dots only if it's not part of the zone index and reset the last length.
if (!zoneIndex) {
dots++
currentLabelLength = 0
}
// if the length of the last section is longer than or equal to 64, it's too long to be a valid domain
if (currentLabelLength >= MAX_LABEL_LENGTH) {
return ReaderNextState.InvalidDomainName
}
}
} else if (seenBracket && (isHex(curr) || curr == ':' || curr == '[' || curr == ']' || curr == '%') &&
!seenCompleteBracketSet
) { // if this is an ipv6 address.
when (curr) {
':' -> {
currentLabelLength = 0
}
'[' -> {
// if we read another '[', we need to restart by re-reading from this bracket instead.
reader.goBack()
return ReaderNextState.InvalidDomainName
}
']' -> {
seenCompleteBracketSet =
true // means that we already have a complete ipv6 address.
zoneIndex =
false // set this back off so that we can keep counting dots after ipv6 is over.
}
'%' -> {
zoneIndex = true
}
else -> {
currentLabelLength++
}
}
numeric = false
buffer.append(curr)
} else if (isAlphaNumeric(curr) || curr == '-' || curr.code >= INTERNATIONAL_CHAR_START) {
// Valid domain name character. Either a-z, A-Z, 0-9, -, or international character
if (seenCompleteBracketSet) {
// covers case of [fe80::]www.google.com
reader.goBack()
done = true
} else {
if (isAllHexSoFar && !isHex(curr)) {
numeric = false
}
// if its not numeric, remember that;
if (!isAllHexSoFar && !isNumeric(curr)) {
numeric = false
}
// append to the states.
buffer.append(curr)
currentLabelLength++
topLevelLength = currentLabelLength
}
} else if (curr == '[' && !seenBracket) {
seenBracket = true
numeric = false
buffer.append(curr)
} else if (curr == '[' && seenCompleteBracketSet) { // Case where [::][ ...
reader.goBack()
done = true
} else if (curr == '%' && reader.canReadChars(2) && isHex(reader.peekChar(0)) &&
isHex(reader.peekChar(1))
) {
// append to the states.
buffer.append(curr)
buffer.append(reader.read())
buffer.append(reader.read())
currentLabelLength += 3
topLevelLength = currentLabelLength
} else {
// invalid character, we are done.
done = true
}
}
// Check the domain name to make sure its ok.
return checkDomainNameValid(ReaderNextState.ValidDomainName, null)
}
fun String.isXn() =
this.length > 3 &&
(this[0] == 'x' || this[0] == 'X') &&
(this[1] == 'n' || this[1] == 'N') &&
this[2] == '-' &&
this[3] == '-'
/**
* Checks the current state of this object and returns if the valid state indicates that the
* object has a valid domain name. If it does, it will return append the last character
* and return the validState specified.
* @param validState The state to return if this check indicates that the dns is ok.
* @param lastChar The last character to add if the domain is ok.
* @return The validState if the domain is valid, else ReaderNextState.InvalidDomainName
*/
private fun checkDomainNameValid(
validState: ReaderNextState?,
lastChar: Char?,
): ReaderNextState? {
var valid = false
// Max domain length is 255 which includes the trailing "."
// most of the time this is not included in the url.
// If the _currentLabelLength is not 0 then the last "." is not included so add it.
// Same with number of labels (or dots including the last)
val lastDotLength =
if (buffer.length > 3 &&
buffer[buffer.length - 3] == '%' &&
buffer[buffer.length - 2] == '2' &&
(buffer[buffer.length - 1] == 'e' || buffer[buffer.length - 1] == 'E')
) {
3
} else {
1
}
val domainLength: Int =
buffer.length - startDomainName + (if (currentLabelLength > 0) lastDotLength else 0)
val dotCount = dots + (if (currentLabelLength > 0) 1 else 0)
if (domainLength >= MAX_DOMAIN_LENGTH || (dotCount > MAX_NUMBER_LABELS)) {
valid = false
} else if (numeric) {
val testDomain = buffer.substring(startDomainName).lowercase()
valid = isValidIpv4(testDomain)
} else if (seenBracket) {
val testDomain = buffer.substring(startDomainName).lowercase()
valid = isValidIpv6(testDomain)
} else if ((currentLabelLength > 0 && dots >= 1) || (dots >= 2 && currentLabelLength == 0)) {
var topStart: Int = buffer.length - topLevelLength
if (currentLabelLength == 0) {
topStart--
}
topStart = max(topStart, 0)
// get the first 4 characters of the top level domain
val topLevelStart =
buffer.substring(topStart, topStart + min(4, buffer.length - topStart))
// There is no size restriction if the top level domain is international (starts with "xn--")
valid =
((topLevelStart.isXn() || (topLevelLength in MIN_TOP_LEVEL_DOMAIN..MAX_TOP_LEVEL_DOMAIN)))
}
if (valid) {
// if it's valid, add the last character (if specified) and return the valid state.
if (lastChar != null) {
buffer.append(lastChar)
}
return validState
}
// Roll back one char if its invalid to handle: "00:41.<br />"
// This gets detected as 41.br otherwise.
reader.goBack()
// return invalid state.
return ReaderNextState.InvalidDomainName
}
/**
* Handles Hexadecimal, octal, decimal, dotted decimal, dotted hex, dotted octal.
* @param testDomain the string we're testing
* @return Returns true if it's a valid ipv4 address
*/
private fun isValidIpv4(testDomain: String): Boolean {
var valid = false
val length: Int = testDomain.length
if (length > 0) {
// handling format without dots. Ex: http://2123123123123/path/a, http://0x8242343/aksdjf
if (dots == 0) {
try {
val value: Long
if (length > 2 && testDomain[0] == '0' && testDomain[1] == 'x') { // hex
// digit must be within ['0', '9'] or ['A', 'F'] or ['a', 'f']
for (c in 2..<length) {
val d: Char = testDomain[c]
if ((d < '0' || (d in ':'..<'A') || (d in 'G'..<'a') || d > 'f')) {
return false
}
}
value = testDomain.substring(2).toLong(16)
} else if (testDomain[0] == '0') { // octal
// digit must be within ['0', '7']
for (c in 1..<length) {
val d: Char = testDomain[c]
if (d !in '0'..'7') {
return false
}
}
value = testDomain.substring(1).toLong(8)
} else { // decimal
// digit must be within ['0', '9']
for (c in 0..<length) {
val d: Char = testDomain[c]
if (d !in '0'..'9') {
return false
}
}
value = testDomain.toLong()
}
valid = value in MIN_NUMERIC_DOMAIN_VALUE..MAX_NUMERIC_DOMAIN_VALUE
} catch (_: NumberFormatException) {
valid = false
}
} else if (dots == 3) {
// Dotted decimal/hex/octal format
val parts: List<String> = splitByDot(testDomain)
valid = true
// check each part of the ip and make sure its valid.
var i = 0
while (i < parts.size && valid) {
val part = parts[i]
val partLen: Int = part.length
if (partLen > 0) {
val parsedNum: String
val base: Int
if (partLen > 2 && part[0] == '0' && part[1] == 'x') { // dotted hex
// digit must be within ['0', '9'] or ['A', 'F'] or ['a', 'f']
for (c in 2..<partLen) {
val d: Char = part[c]
if ((d < '0' || (d in ':'..<'A') || (d in 'G'..<'a') || d > 'f')) {
return false
}
}
parsedNum = part.substring(2)
base = 16
} else if (part[0] == '0') { // dotted octal
// digit must be within ['0', '7']
for (c in 1..<partLen) {
val d: Char = part[c]
if (d !in '0'..'7') {
return false
}
}
parsedNum = part.substring(1)
base = 8
} else { // dotted decimal
// digit must be within ['0', '9']
for (c in 0..<partLen) {
val d: Char = part[c]
if (d !in '0'..'9') {
return false
}
}
parsedNum = part
base = 10
}
val section =
if (parsedNum.isEmpty()) {
0
} else {
try {
parsedNum.toInt(base)
} catch (_: NumberFormatException) {
return false
}
}
if (section !in MIN_IP_PART..MAX_IP_PART) {
valid = false
}
} else {
valid = false
}
i++
}
}
}
return valid
}
/**
* Sees that there's an open "[", and is now checking for ":"'s and stopping when there is a ']' or invalid character.
* Handles ipv4 formatted ipv6 addresses, zone indices, truncated notation.
* @return Returns true if it is a valid ipv6 address
*/
private fun isValidIpv6(testDomain: String): Boolean {
val domainArray = testDomain.toCharArray()
// Return false if we don't see [....]
// or if we only have '[]'
// or if we detect [:8000: ...]; only [::8000: ...] is okay
if (
domainArray.size < 3 ||
domainArray[domainArray.size - 1] != ']' ||
domainArray[0] != '[' ||
(domainArray[1] == ':' && domainArray[2] != ':')
) {
return false
}
var numSections = 1
var hexDigits = 0
var prevChar = 0.toChar()
// used to check ipv4 addresses at the end of ipv6 addresses.
val lastSection = StringBuilder()
var hexSection = true
// If we see a '%'. Example: http://[::ffff:0xC0.0x00.0x02.0xEB%251]
var zoneIndiceMode = false
// If doubleColonFlag is true, that means we've already seen one "::"; we're not allowed to have more than one.
var doubleColonFlag = false
var index = 0
while (index < domainArray.size) {
when (domainArray[index]) {
'[' -> {}
'%', ']' -> {
var out = false
if (domainArray[index] == '%') {
// see if there's a urlencoded dot
if (domainArray.size - index >= 2 && domainArray[index + 1] == '2' && domainArray[index + 2] == 'e') {
lastSection.append("%2e")
index += 2
hexSection = false
out = true
}
if (!out) zoneIndiceMode = true
}
if (!out) {
if (!hexSection && (!zoneIndiceMode || domainArray[index] == '%')) {
if (isValidIpv4(lastSection.toString())) {
numSections++ // ipv4 takes up 2 sections.
} else {
return false
}
}
}
}
':' -> {
if (prevChar == ':') {
if (doubleColonFlag) { // only allowed to have one "::" in an ipv6 address.
return false
}
doubleColonFlag = true
}
// This means that we reached invalid characters in the previous section
if (!hexSection) {
return false
}
hexSection = true // reset hex to true
hexDigits = 0 // reset count for hex digits
numSections++
lastSection.deleteRange(0, lastSection.length) // clear last section
}
else -> {
if (zoneIndiceMode) {
if (!isUnreserved(domainArray[index])) {
return false
}
} else {
lastSection.append(domainArray[index]) // collect our possible ipv4 address
if (hexSection && isHex(domainArray[index])) {
hexDigits++
} else {
hexSection = false // non hex digit.
}
}
}
}
if (hexDigits > 4 || numSections > 8) {
return false
}
prevChar = domainArray[index]
index++
}
// numSections != 1 checks for things like: [adf]
// If there are more than 8 sections for the address or there isn't a double colon, then it's invalid.
return numSections != 1 && (numSections >= 8 || doubleColonFlag)
}
companion object {
/**
* The minimum length of a ascii based top level domain.
*/
private const val MIN_TOP_LEVEL_DOMAIN = 2
/**
* The maximum length of a ascii based top level domain.
*/
private const val MAX_TOP_LEVEL_DOMAIN = 22
/**
* The maximum number that the url can be in a url that looks like:
* http://123123123123/path
*/
private const val MAX_NUMERIC_DOMAIN_VALUE = 4294967295L
/**
* The minimum number the url can be in a url that looks like:
* http://123123123123/path
*/
private const val MIN_NUMERIC_DOMAIN_VALUE = 16843008L
/**
* If the domain name is an ip address, for each part of the address, whats the minimum value?
*/
private const val MIN_IP_PART = 0
/**
* If the domain name is an ip address, for each part of the address, whats the maximum value?
*/
private const val MAX_IP_PART = 255
/**
* The start of the utf character code table which indicates that this character is an international character.
* Everything below this value is either a-z,A-Z,0-9 or symbols that are not included in domain name.
*/
private const val INTERNATIONAL_CHAR_START = 192
/**
* The maximum length of each label in the domain name.
*/
private const val MAX_LABEL_LENGTH = 64
/**
* The maximum number of labels in a single domain name.
*/
private const val MAX_NUMBER_LABELS = 127
/**
* The maximum domain name length.
*/
private const val MAX_DOMAIN_LENGTH = 255
}
}
@@ -0,0 +1,101 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector.detection
import com.vitorpamplona.quartz.utils.urldetector.detection.CharUtils.isWhiteSpace
/**
* Class used to read a text input character by character. This also gives the ability to backtrack.
*/
class InputTextReader(
content: String,
) {
/**
* The content to read.
*/
private val content: CharArray = content.toCharArray()
/**
* The current position in the content we are looking at.
*/
var position: Int = 0
private set
/**
* Reads a single char from the content stream and increments the index.
* @return The next available character.
*/
fun read(): Char {
val chr = content[this.position++]
return if (isWhiteSpace(chr)) ' ' else chr
}
/**
* Peeks at the next number of chars and returns as a string without incrementing the current index.
* @param str The string to compare to
*/
fun peekEquals(str: String): Boolean {
if (position + str.length > content.size) return false
for (i in str.indices) {
if (content[position + i] != str[i]) return false
}
return true
}
/**
* Gets the character in the array offset by the current index.
* @param offset The number of characters to offset.
* @return The character at the location of the index plus the provided offset.
*/
fun peekChar(offset: Int): Char {
if (!canReadChars(offset)) {
throw IllegalArgumentException("Index out of bounds")
}
return content[this.position + offset]
}
/**
* Returns true if the reader has more the specified number of chars.
* @param numberChars The number of chars to see if we can read.
* @return True if we can read this number of chars, else false.
*/
fun canReadChars(numberChars: Int): Boolean = content.size >= this.position + numberChars
/**
* Checks if the current stream is at the end.
* @return True if the stream is at the end and no more can be read.
*/
fun eof(): Boolean = content.size <= this.position
/**
* Moves the index to the specified position.
* @param position The position to set the index to.
*/
fun seek(position: Int) {
this.position = position
}
/**
* Goes back a single character.
*/
fun goBack() = this.position--
}
@@ -0,0 +1,597 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector.detection
import com.vitorpamplona.quartz.utils.urldetector.Url
import com.vitorpamplona.quartz.utils.urldetector.UrlMarker
import com.vitorpamplona.quartz.utils.urldetector.UrlPart
import kotlin.math.max
class UrlDetector(
content: String,
) {
/**
* The input stream to read.
*/
private val reader: InputTextReader = InputTextReader(content)
/**
* Buffer to store temporary urls inside of.
*/
private val buffer = StringBuilder()
/**
* Has the scheme been found in this iteration?
*/
private var hasScheme = false
/**
* If the first character in the url is a quote, then look for matching quote at the end.
*/
private var quoteStart = false
/**
* Stores the found urls.
*/
private val urlList: ArrayList<Url> = ArrayList<Url>()
/**
* Keeps track of certain indices to create a Url object.
*/
private var currentUrlMarker: UrlMarker = UrlMarker()
/**
* The states to use to continue writing or not.
*/
enum class ReadEndState {
/**
* The current url is valid.
*/
ValidUrl,
/**
* The current url is invalid.
*/
InvalidUrl,
}
/**
* Detects the urls and returns a list of detected url strings.
* @return A list with detected urls.
*/
fun detect(): List<Url> {
readDefault()
return urlList
}
/**
* The default input reader which looks for specific flags to start detecting the url.
*/
private fun readDefault() {
// Keeps track of the number of characters read to be able to later cut out the domain name.
var length = 0
var position = 0
// until end of string read the contents
while (!reader.eof()) {
// read the next char to process.
when (val curr = reader.read()) {
' ' -> {
buffer.append(curr)
readEnd(ReadEndState.InvalidUrl)
length = 0
}
'%' -> {
if (reader.canReadChars(2)) {
if (reader.peekEquals("3a") || reader.peekEquals("3A")) {
buffer.append(curr)
buffer.append(reader.read())
buffer.append(reader.read())
length = processColon(length)
} else if (CharUtils.isHex(reader.peekChar(0)) && CharUtils.isHex(reader.peekChar(1))) {
buffer.append(curr)
buffer.append(reader.read())
buffer.append(reader.read())
if (!readDomainName(buffer.substring(length))) {
readEnd(ReadEndState.InvalidUrl)
}
length = 0
}
}
}
'\u3002', '\uFF0E', '\uFF61', '.' -> {
buffer.append(curr)
val domain = buffer.substring(length)
if (!readDomainName(domain)) {
readEnd(ReadEndState.InvalidUrl)
}
length = 0
}
'@' -> {
if (buffer.isNotEmpty()) {
currentUrlMarker.setIndex(UrlPart.USERNAME_PASSWORD, length)
buffer.append(curr)
if (!readDomainName(null)) {
readEnd(ReadEndState.InvalidUrl)
}
length = 0
}
}
'[' -> {
val beginning = reader.position
// if it doesn't have a scheme, clear the buffer.
if (!hasScheme) {
buffer.clear()
}
buffer.append(curr)
if (!readDomainName(buffer.substring(length))) {
// if we didn't find an ipv6 address, then check inside the brackets for urls
readEnd(ReadEndState.InvalidUrl)
reader.seek(beginning)
}
length = 0
}
'/' -> {
// "/" was found, then we either read a scheme, or if we already read a scheme, then
// we are reading a url in the format http://123123123/asdf
if (hasScheme) {
// we already have the scheme, so then we already read:
// http://something/ <- if something is all numeric then its a valid url.
// OR we are searching for single level domains. We have buffer length > 1 condition
// to weed out infinite backtrack in cases of html5 roots
// unread this "/" and continue to check the domain name starting from the beginning of the domain
reader.goBack()
if (!readDomainName(buffer.substring(length))) {
readEnd(ReadEndState.InvalidUrl)
}
length = 0
} else {
// we don't have a scheme already, then clear state, then check for html5 root such as: "//google.com/"
// remember the state of the quote when clearing state just in case its "//google.com" so its not cleared.
readEnd(ReadEndState.InvalidUrl)
buffer.append(curr)
hasScheme = readHtml5Root()
length = buffer.length
}
}
':' -> {
// add the ":" to the url and check for scheme/username
buffer.append(curr)
length = processColon(length)
}
else -> {
buffer.append(curr)
}
}
if (position == reader.position) {
// we haven't made any progress, advance by one char
reader.read()
}
position = reader.position
}
}
/**
* We found a ":" and is now trying to read either scheme, username/password
* @param length first index of the previous part (could be beginning of the buffer, beginning of the username/password, or beginning
* @return new index of where the domain starts
*/
private fun processColon(length: Int): Int {
var length = length
if (hasScheme) {
// read it as username/password if it has scheme
if (!readUserPass(length)) {
// unread the ":" so that the domain reader can process it
reader.goBack()
// Check buffer length before clearing it; set length to 0 if buffer is empty
if (buffer.length > 0) {
buffer.deleteRange(buffer.length - 1, buffer.length)
} else {
length = 0
}
val backtrackOnFail: Int = reader.position - buffer.length + length
if (!readDomainName(buffer.substring(length))) {
// go back to length location and restart search
reader.seek(backtrackOnFail)
readEnd(ReadEndState.InvalidUrl)
}
length = 0
} else {
length = 0
}
} else if (readScheme() && buffer.isNotEmpty()) {
hasScheme = true
length = buffer.length // set length to be right after the scheme
} else {
readEnd(ReadEndState.InvalidUrl)
length = 0
}
return length
}
/**
* Checks if the url is in the format:
* //google.com/static/js.js
* @return True if the url is in this format and was matched correctly.
*/
private fun readHtml5Root(): Boolean {
// end of input then go away.
if (reader.eof()) {
return false
}
// read the next character. If its // then return true.
val curr = reader.read()
if (curr == '/') {
buffer.append(curr)
return true
} else {
// if its not //, then go back and reset by 1 character.
reader.goBack()
readEnd(ReadEndState.InvalidUrl)
}
return false
}
/**
* Reads the scheme and allows returns true if the scheme is http(s?):// or ftp(s?)://
* @return True if the scheme was found, else false.
*/
private fun readScheme(): Boolean {
val originalLength: Int = buffer.length
var numSlashes = 0
while (!reader.eof()) {
val curr = reader.read()
// if we match a slash, look for a second one.
if (curr == '/') {
buffer.append(curr)
if (numSlashes == 1) {
// return only if its an approved protocol. This can be expanded to allow others
val schemeStartIndex: Int = findValidSchemeStartIndex(buffer.toString())
if (schemeStartIndex >= 0) {
buffer.deleteRange(0, schemeStartIndex)
currentUrlMarker.setIndex(UrlPart.SCHEME, 0)
return true
} else {
return false
}
}
numSlashes++
} else if (curr == ' ') {
// if we find a space or end of input, then nothing found.
buffer.append(curr)
return false
} else if (curr == '[') { // if we're starting to see an ipv6 address
reader.goBack() // unread the '[', so that we can start looking for ipv6
return false
} else if (originalLength > 0 || numSlashes > 0 || !CharUtils.isAlpha(curr)) {
// if it's not a character a-z or A-Z then assume we aren't matching scheme, but instead
// matching username and password.
reader.goBack()
return readUserPass(0)
}
}
return false
}
private fun findValidSchemeStartIndex(optionalScheme: String): Int {
val optionalSchemeLowercase = optionalScheme.lowercase()
return VALID_SCHEMES
.filter(optionalSchemeLowercase::endsWith)
.map(optionalSchemeLowercase::lastIndexOf)
.firstOrNull() ?: -1
}
/**
* Reads the input and looks for a username and password.
* Handles:
* http://username:password@...
* @param beginningOfUsername Index of the buffer of where the username began
* @return True if a valid username and password was found.
*/
private fun readUserPass(beginningOfUsername: Int): Boolean {
// The start of where we are.
val start: Int = buffer.length
// keep looping until "done"
var done = false
// if we had a dot in the input, then it might be a domain name and not a username and password.
var rollback = false
while (!done && !reader.eof()) {
val curr = reader.read()
// if we hit this, then everything is ok and we are matching a domain name.
if (curr == '@') {
buffer.append(curr)
currentUrlMarker.setIndex(UrlPart.USERNAME_PASSWORD, beginningOfUsername)
return readDomainName("")
} else if (CharUtils.isDot(curr) || curr == '[') {
// everything is still ok, just remember that we found a dot or '[' in case we might need to backtrack
buffer.append(curr)
rollback = true
} else if (curr == '#' || curr == ' ' || curr == '/') {
// one of these characters indicates we are invalid state and should just return.
rollback = true
done = true
} else {
// all else, just append character assuming its ok so far.
buffer.append(curr)
}
}
if (rollback) {
// got to here, so there is no username and password. (We didn't find a @)
val distance: Int = buffer.length - start
buffer.deleteRange(start, buffer.length)
val currIndex: Int = max(reader.position - distance - (if (done) 1 else 0), 0)
reader.seek(currIndex)
return false
} else {
return readEnd(ReadEndState.InvalidUrl)
}
}
/**
* Try to read the current string as a domain name
* @param current The current string used.
* @return Whether the domain is valid or not.
*/
private fun readDomainName(current: String?): Boolean {
val hostIndex: Int =
if (current == null) buffer.length else buffer.length - current.length
currentUrlMarker.setIndex(UrlPart.HOST, hostIndex)
// create the domain name reader and specify the handler that will be called when a quote character
// or something is found.
val reader = DomainNameReader(reader, buffer, current)
// Try to read the dns and act on the response.
val state = reader.readDomainName()
return when (state) {
DomainNameReader.ReaderNextState.ValidDomainName -> {
readEnd(ReadEndState.ValidUrl)
}
DomainNameReader.ReaderNextState.ReadFragment -> {
readFragment()
}
DomainNameReader.ReaderNextState.ReadPath -> {
readPath()
}
DomainNameReader.ReaderNextState.ReadPort -> {
readPort()
}
DomainNameReader.ReaderNextState.ReadQueryString -> {
readQueryString()
}
DomainNameReader.ReaderNextState.ReadUserPass -> {
val host: Int = currentUrlMarker.indexOf(UrlPart.HOST)
currentUrlMarker.unsetIndex(UrlPart.HOST)
readUserPass(host)
}
else -> {
false
}
}
}
/**
* Reads the fragments which is the part of the url starting with #
* @return If a valid fragment was read true, else false.
*/
private fun readFragment(): Boolean {
currentUrlMarker.setIndex(UrlPart.FRAGMENT, buffer.length - 1)
while (!reader.eof()) {
val curr = reader.read()
// if it's the end or space, then a valid url was read.
if (curr == ' ') {
return readEnd(ReadEndState.ValidUrl)
} else {
// otherwise keep appending.
buffer.append(curr)
}
}
// if we are here, anything read is valid.
return readEnd(ReadEndState.ValidUrl)
}
/**
* Try to read the query string.
* @return True if the query string was valid.
*/
private fun readQueryString(): Boolean {
currentUrlMarker.setIndex(UrlPart.QUERY, buffer.length - 1)
while (!reader.eof()) {
val curr = reader.read()
if (curr == '#') { // fragment
buffer.append(curr)
return readFragment()
} else if (curr == ' ') {
// end of query string
return readEnd(ReadEndState.ValidUrl)
} else { // all else add to buffer.
buffer.append(curr)
}
}
// a valid url was read.
return readEnd(ReadEndState.ValidUrl)
}
/**
* Try to read the port of the url.
* @return True if a valid port was read.
*/
private fun readPort(): Boolean {
currentUrlMarker.setIndex(UrlPart.PORT, buffer.length)
// The length of the port read.
var portLen = 0
while (!reader.eof()) {
// read the next one and remember the length
val curr = reader.read()
portLen++
if (curr == '/') {
// continue to read path
buffer.append(curr)
return readPath()
} else if (curr == '?') {
// continue to read query string
buffer.append(curr)
return readQueryString()
} else if (curr == '#') {
// continue to read fragment.
buffer.append(curr)
return readFragment()
} else if (!CharUtils.isNumeric(curr)) {
// if we got here, then what we got so far is a valid url. don't append the current character.
reader.goBack()
// no port found; it was something like google.com:hello.world
if (portLen == 1) {
// remove the ":" from the end.
buffer.deleteRange(buffer.length - 1, buffer.length)
}
currentUrlMarker.unsetIndex(UrlPart.PORT)
return readEnd(ReadEndState.ValidUrl)
} else {
// this is a valid character in the port string.
buffer.append(curr)
}
}
// found a correct url
return readEnd(ReadEndState.ValidUrl)
}
/**
* Tries to read the path
* @return True if the path is valid.
*/
private fun readPath(): Boolean {
currentUrlMarker.setIndex(UrlPart.PATH, buffer.length - 1)
while (!reader.eof()) {
// read the next char
val curr = reader.read()
if (curr == ' ') {
// if end of state and we got here, then the url is valid.
return readEnd(ReadEndState.ValidUrl)
}
// append the char
buffer.append(curr)
// now see if we move to another state.
if (curr == '?') {
// if ? read query string
return readQueryString()
} else if (curr == '#') {
// if # read the fragment
return readFragment()
}
}
// end of input then this url is good.
return readEnd(ReadEndState.ValidUrl)
}
/**
* The url has been read to here. Remember the url if its valid, and reset state.
* @param state The state indicating if this url is valid. If its valid it will be added to the list of urls.
* @return True if the url was valid.
*/
private fun readEnd(state: ReadEndState?): Boolean {
// if the url is valid and greater then 0
if (state == ReadEndState.ValidUrl && buffer.isNotEmpty()) {
// get the last character. if its a quote, cut it off.
val len: Int = buffer.length
if (quoteStart && buffer[len - 1] == '\"') {
buffer.deleteRange(len - 1, len)
}
// Add the url to the list of good urls.
if (buffer.isNotEmpty()) {
currentUrlMarker.originalUrl = buffer.toString()
urlList.add(currentUrlMarker.createUrl())
}
}
// clear out the buffer.
buffer.deleteRange(0, buffer.length)
// reset the state of internal objects.
quoteStart = false
hasScheme = false
currentUrlMarker = UrlMarker()
// return true if valid.
return state == ReadEndState.ValidUrl
}
companion object {
private val VALID_SCHEMES: List<String> =
listOf(
"http://",
"https://",
"ftp://",
"ftps://",
"ws://",
"wss://",
// "nostr:",
// "blossom:",
)
}
}
@@ -20,6 +20,7 @@
*/
package com.vitorpamplona.quartz.nip64Chess
import com.vitorpamplona.quartz.nip31Alts.alt
import com.vitorpamplona.quartz.nip64Chess.game.ChessGameEvent
import kotlin.test.Test
import kotlin.test.assertEquals
@@ -83,7 +84,7 @@ class ChessGameEventTest {
sig = "test_sig",
)
assertEquals(customAltText, testEvent.altText(), "Alt text should be extractable from tags")
assertEquals(customAltText, testEvent.alt(), "Alt text should be extractable from tags")
}
@Test
@@ -98,7 +99,7 @@ class ChessGameEventTest {
sig = "test_sig",
)
assertEquals(null, testEvent.altText(), "Should return null when no alt tag present")
assertEquals(null, testEvent.alt(), "Should return null when no alt tag present")
}
@Test
@@ -0,0 +1,322 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector
import kotlin.test.Test
import kotlin.test.assertEquals
class UrlMarkerTest {
fun testUrlMarker(
testString: String,
scheme: String?,
username: String?,
password: String?,
host: String?,
port: Int,
path: String?,
query: String?,
fragment: String?,
indices: IntArray,
) {
val urlMarker = UrlMarker()
urlMarker.originalUrl = testString
urlMarker.setIndices(indices)
val url = urlMarker.createUrl()
assertEquals(url.host, host, "host, " + testString)
assertEquals(url.path, path, "path, " + testString)
assertEquals(url.scheme, scheme, "scheme, " + testString)
assertEquals(url.username, username, "username, " + testString)
assertEquals(url.password, password, "password, " + testString)
assertEquals(url.port, port, "port, " + testString)
assertEquals(url.query, query, "query, " + testString)
assertEquals(url.fragment, fragment, "fragment, " + testString)
}
@Test
fun test1() =
testUrlMarker(
"hello@hello.com",
"https",
"hello",
"",
"hello.com",
443,
"/",
"",
"",
intArrayOf(-1, 0, 6, -1, -1, -1, -1),
)
@Test
fun test2() =
testUrlMarker(
"http://hello@hello.com",
"http",
"hello",
"",
"hello.com",
80,
"/",
"",
"",
intArrayOf(0, 7, 13, -1, -1, -1, -1),
)
@Test
fun test3() =
testUrlMarker(
"hello@hello.com",
"https",
"hello",
"",
"hello.com",
443,
"/",
"",
"",
intArrayOf(-1, 0, 6, -1, -1, -1, -1),
)
@Test
fun test4() =
testUrlMarker(
"https://user@google.com/h?hello=w#abc",
"https",
"user",
"",
"google.com",
443,
"/h",
"?hello=w",
"#abc",
intArrayOf(0, 8, 13, -1, 23, 25, 33),
)
@Test
fun test5() =
testUrlMarker(
"www.booopp.com:20#fa",
"https",
"",
"",
"www.booopp.com",
20,
"/",
"",
"#fa",
intArrayOf(-1, -1, 0, 15, -1, -1, 17),
)
@Test
fun test6() =
testUrlMarker(
"www.yahooo.com:20?fff#aa",
"https",
"",
"",
"www.yahooo.com",
20,
"/",
"?fff",
"#aa",
intArrayOf(-1, -1, 0, 15, -1, 17, 21),
)
@Test
fun test7() =
testUrlMarker(
"www.google.com#fa",
"https",
"",
"",
"www.google.com",
443,
"/",
"",
"#fa",
intArrayOf(-1, -1, 0, -1, -1, -1, 14),
)
@Test
fun test8() =
testUrlMarker(
"www.google.com?3fd#fa",
"https",
"",
"",
"www.google.com",
443,
"/",
"?3fd",
"#fa",
intArrayOf(-1, -1, 0, -1, -1, 14, 18),
)
@Test
fun test9() =
testUrlMarker(
"//www.google.com/",
"",
"",
"",
"www.google.com",
-1,
"/",
"",
"",
intArrayOf(-1, -1, 2, -1, 16, -1, -1),
)
@Test
fun test10() =
testUrlMarker(
"http://www.google.com/",
"http",
"",
"",
"www.google.com",
80,
"/",
"",
"",
intArrayOf(0, -1, 7, -1, 21, -1, -1),
)
@Test
fun test11() =
testUrlMarker(
"ftp://whosdere:me@google.com/",
"ftp",
"whosdere",
"me",
"google.com",
21,
"/",
"",
"",
intArrayOf(0, 6, 18, -1, 28, -1, -1),
)
@Test
fun test12() =
testUrlMarker(
"ono:doope@fb.net:9090/dhdh",
"https",
"ono",
"doope",
"fb.net",
9090,
"/dhdh",
"",
"",
intArrayOf(-1, 0, 10, 17, 21, -1, -1),
)
@Test
fun test13() =
testUrlMarker(
"ono:a@fboo.com:90/dhdh/@1234",
"https",
"ono",
"a",
"fboo.com",
90,
"/dhdh/@1234",
"",
"",
intArrayOf(-1, 0, 6, 15, 17, -1, -1),
)
@Test
fun test14() =
testUrlMarker(
"fbeoo.net:990/dhdeh/@1234",
"https",
"",
"",
"fbeoo.net",
990,
"/dhdeh/@1234",
"",
"",
intArrayOf(-1, -1, 0, 10, 13, -1, -1),
)
@Test
fun test15() =
testUrlMarker(
"fbeoo:@boop.com/dhdeh/@1234?aj=r",
"https",
"fbeoo",
"",
"boop.com",
443,
"/dhdeh/@1234",
"?aj=r",
"",
intArrayOf(-1, 0, 7, -1, 15, 27, -1),
)
@Test
fun test16() =
testUrlMarker(
"bloop:@noooo.com/doop/@1234",
"https",
"bloop",
"",
"noooo.com",
443,
"/doop/@1234",
"",
"",
intArrayOf(-1, 0, 7, -1, 16, -1, -1),
)
@Test
fun test17() =
testUrlMarker(
"bah.com/lala/@1234/@dfd@df?@dsf#ono",
"https",
"",
"",
"bah.com",
443,
"/lala/@1234/@dfd@df",
"?@dsf",
"#ono",
intArrayOf(-1, -1, 0, -1, 7, 26, 31),
)
@Test
fun test18() =
testUrlMarker(
"https://dewd:dood@www.google.com:20/?why=is&this=test#?@Sdsf",
"https",
"dewd",
"dood",
"www.google.com",
20,
"/",
"?why=is&this=test",
"#?@Sdsf",
intArrayOf(0, 8, 18, 33, 35, 36, 53),
)
}
@@ -0,0 +1,121 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector.detection
import kotlin.test.Test
import kotlin.test.assertContentEquals
import kotlin.test.assertFalse
import kotlin.test.assertTrue
class CharUtilsTest {
@Test
fun testCharUtilsIsHex() {
val arr = charArrayOf('a', 'A', '0', '9')
for (a in arr) {
assertTrue(CharUtils.isHex(a))
}
val arr2 = charArrayOf('~', ';', 'Z', 'g')
for (a in arr2) {
assertFalse(CharUtils.isHex(a))
}
}
@Test
fun testCharUtilsIsNumeric() {
val arr = charArrayOf('0', '4', '6', '9')
for (a in arr) {
assertTrue(CharUtils.isNumeric(a))
}
val arr2 = charArrayOf('a', '~', 'A', 0.toChar())
for (a in arr2) {
assertFalse(CharUtils.isNumeric(a))
}
}
@Test
fun testCharUtilsIsAlpha() {
val arr = charArrayOf('a', 'Z', 'f', 'X')
for (a in arr) {
assertTrue(CharUtils.isAlpha(a))
}
val arr2 = charArrayOf('0', '9', '[', '~')
for (a in arr2) {
assertFalse(CharUtils.isAlpha(a))
}
}
@Test
fun testCharUtilsIsAlphaNumeric() {
val arr = charArrayOf('a', 'G', '3', '9')
for (a in arr) {
assertTrue(CharUtils.isAlphaNumeric(a))
}
val arr2 = charArrayOf('~', '-', '_', '\n')
for (a in arr2) {
assertFalse(CharUtils.isAlphaNumeric(a))
}
}
@Test
fun testCharUtilsIsUnreserved() {
val arr = charArrayOf('-', '.', 'a', '9', 'Z', '_', 'f')
for (a in arr) {
assertTrue(CharUtils.isUnreserved(a))
}
val arr2 = charArrayOf(' ', '!', '(', '\n')
for (a in arr2) {
assertFalse(CharUtils.isUnreserved(a))
}
}
@Test
fun testSplitByDot() {
val stringsToSplit =
listOf(
"192.168.1.1",
"..",
"192%2e168%2e1%2e1",
"asdf",
"192.39%2e1%2E1",
"as\uFF61awe.a3r23.lkajsf0ijr....",
"%2e%2easdf",
"sdoijf%2e",
"ksjdfh.asdfkj.we%2",
"0xc0%2e0x00%2e0x02%2e0xeb",
"",
)
val regex = "[\\.\u3002\uFF0E\uFF61]|%2e|%2E".toRegex()
stringsToSplit.forEach { stringToSplit ->
assertContentEquals(
stringToSplit.split(regex),
CharUtils.splitByDot(stringToSplit),
"Splitting $stringToSplit",
)
}
}
}
@@ -0,0 +1,76 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector.detection
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertFalse
import kotlin.test.assertTrue
class InputTextReaderTest {
@Test
fun testSimpleRead() {
val reader = InputTextReader(CONTENT)
for (i in 0..<CONTENT.length) {
assertEquals(reader.read(), CONTENT[i])
}
}
@Test
fun testEOF() {
val reader = InputTextReader(CONTENT)
for (i in 0..<CONTENT.length - 1) {
reader.read()
}
assertFalse(reader.eof())
reader.read()
assertTrue(reader.eof())
}
@Test
fun testGoBack() {
val reader = InputTextReader(CONTENT)
assertEquals(reader.read(), CONTENT[0])
reader.goBack()
assertEquals(reader.read(), CONTENT[0])
assertEquals(reader.read(), CONTENT[1])
assertEquals(reader.read(), CONTENT[2])
reader.goBack()
reader.goBack()
assertEquals(reader.read(), CONTENT[1])
assertEquals(reader.read(), CONTENT[2])
}
@Test
fun testSeek() {
val reader = InputTextReader(CONTENT)
reader.seek(4)
assertEquals(reader.read(), CONTENT[4])
reader.seek(1)
assertEquals(reader.read(), CONTENT[1])
}
companion object {
private val CONTENT = "HELLO WORLD"
}
}
@@ -0,0 +1,643 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.utils.urldetector.detection
import com.vitorpamplona.quartz.utils.urldetector.Url
import kotlin.test.Test
import kotlin.test.assertEquals
class UriDetectionTest {
@Test
fun testBasicString() {
runTest("hello world")
}
@Test
fun testBasicDetect() {
runTest("this is a link: www.google.com", "www.google.com")
}
@Test
fun testSimple() {
runTest(
"http://www.linkedin.com/vshlos",
"http://www.linkedin.com/vshlos",
)
}
@Test
fun testEmailAndNormalUrl() {
runTest(
"my email is vshlosbe@linkedin.com and my site is http://www.linkedin.com/vshlos",
"vshlosbe@linkedin.com",
"http://www.linkedin.com/vshlos",
)
}
@Test
fun testTwoBasicUrls() {
runTest(
"the url google.com is a lot better then www.google.com.",
"google.com",
"www.google.com.",
)
}
@Test
fun testLongUrl() {
runTest(
"google.com.google.com is kind of a valid url",
"google.com.google.com",
)
}
@Test
fun testInternationalUrls() {
runTest(
"this is an international domain: http://\u043F\u0440\u0438\u043c\u0435\u0440.\u0438\u0441\u043f\u044b" +
"\u0442\u0430\u043d\u0438\u0435 so is this: \u4e94\u7926\u767c\u5c55.\u4e2d\u570b.",
"http://\u043F\u0440\u0438\u043c\u0435\u0440.\u0438\u0441\u043f\u044b\u0442\u0430\u043d\u0438\u0435",
"\u4e94\u7926\u767c\u5c55.\u4e2d\u570b.",
)
}
@Test
fun testDomainWithUsernameAndPassword() {
runTest(
"domain with username is http://username:password@www.google.com/site/1/2",
"http://username:password@www.google.com/site/1/2",
)
}
@Test
fun testFTPWithUsernameAndPassword() {
runTest(
"ftp with username is ftp://username:password@www.google.com",
"ftp://username:password@www.google.com",
)
}
@Test
fun testUncommonFormatUsernameAndPassword() {
runTest(
"weird url with username is username:password@www.google.com",
"username:password@www.google.com",
)
}
@Test
fun testEmailAndLinkWithUserPass() {
runTest(
"email and username is hello@test.google.com or hello@www.google.com hello:password@www.google.com",
"hello@test.google.com",
"hello@www.google.com",
"hello:password@www.google.com",
)
}
@Test
fun testWrongSpacingInSentence() {
runTest(
"I would not like to work at salesforce.com, it looks like a crap company.and not cool!",
"salesforce.com",
"company.and",
)
}
@Test
fun testNumbersAreNotDetected() {
// make sure pure numbers don't work, but domains with numbers do.
runTest("Do numbers work? such as 3.1415 or 4.com", "4.com")
}
@Test
fun testNewLinesAndTabsAreDelimiters() {
runTest(
"Do newlines and tabs break? google.com/hello/\nworld www.yahoo.com\t/stuff/ yahoo.com/\thello news.ycombinator.com\u0000/hello world",
"google.com/hello/",
"www.yahoo.com",
"yahoo.com/",
"news.ycombinator.com",
)
}
@Test
fun testIpAddressFormat() {
runTest(
"How about IP addresses? fake: 1.1.1 1.1.1.1.1 0.0.0.256 255.255.255.256 real: 1.1.1.1 192.168.10.1 1.1.1.1.com 255.255.255.255",
"1.1.1.1",
"192.168.10.1",
"1.1.1.1.com",
"255.255.255.255",
)
}
@Test
fun testNumericIpAddress() {
runTest(
"http://3232235521/helloworld",
"http://3232235521/helloworld",
)
}
@Test
fun testNumericIpAddressWithPort() {
runTest(
"http://3232235521:8080/helloworld",
"http://3232235521:8080/helloworld",
)
}
@Test
fun testDomainAndLabelSizeConstraints() {
// Really long addresses testing rules about total length of domain name and number of labels in a domain and size of each label.
runTest(
(
"This will work: 1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.a.b.c.d.e.ly " +
"This will not work: 1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.a.b.c.d.e.f.ly " +
"This should as well: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa.bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb.ccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc.dddddddddddddddddddddddddddddddddddddddddddddddddddddd.bit.ly " +
"But this wont: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa.bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb.ccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc.dddddddddddddddddddddddddddddddddddddddddddddddddddddd.bit.ly.dbl.spamhaus.org"
),
"1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.1.2.3.4.5.6.7.8.9.0.a.b.c.d.e.ly",
"aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa.bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb.ccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc.dddddddddddddddddddddddddddddddddddddddddddddddddddddd.bit.ly",
)
}
@Test
fun testIncorrectParsingHtmlWithBadOptions() {
runTest(
"<a href=\"http://www.google.com/\">google.com</a>",
"http://www.google.com/\">google.com</a>",
)
}
@Test
fun testNonStandardDots() {
runTest(
"www\u3002google\u3002com username:password@www\uFF0Eyahoo\uFF0Ecom http://www\uFF61facebook\uFF61com http://192\u3002168\uFF0E0\uFF611/",
"www\u3002google\u3002com",
"username:password@www\uFF0Eyahoo\uFF0Ecom",
"http://www\uFF61facebook\uFF61com",
"http://192\u3002168\uFF0E0\uFF611/",
)
}
@Test
fun testNonStandardDotsBacktracking() {
runTest("\u9053 \u83dc\u3002\u3002\u3002\u3002")
}
@Test
fun testBacktrackingStrangeFormats() {
runTest(
"http:http:http://www.google.com www.www:yahoo.com yahoo.com.br hello.hello..hello.com",
"http://www.google.com",
"www.www",
"yahoo.com",
"yahoo.com.br",
"hello.hello.",
"hello.com",
)
}
@Test
fun testBacktrackingUsernamePassword() {
runTest("check out my url:www.google.com", "www.google.com")
runTest("check out my url:www.google.com ", "www.google.com")
}
@Test
fun testBacktrackingEmptyDomainName() {
runTest("check out my http:///hello")
runTest("check out my http://./hello")
}
@Test
fun testDoubleScheme() {
runTest("http://http://")
runTest("hello http://http://")
}
@Test
fun testMultipleSchemes() {
runTest("http://http://www.google.com", "http://www.google.com")
runTest(
"make sure it's right here http://http://www.google.com",
"http://www.google.com",
)
runTest(
"http://http://http://www.google.com",
"http://www.google.com",
)
runTest(
"make sure it's right here http://http://http://www.google.com",
"http://www.google.com",
)
runTest(
"http://ftp://https://www.google.com",
"https://www.google.com",
)
runTest(
"make sure its right here http://ftp://https://www.google.com",
"https://www.google.com",
)
}
@Test
fun testDottedHexIpAddress() {
runTest(
"http://0xc0.0x00.0xb2.0xEB",
"http://0xc0.0x00.0xb2.0xEB",
)
runTest(
"http://0xc0.0x0.0xb2.0xEB",
"http://0xc0.0x0.0xb2.0xEB",
)
runTest(
"http://0x000c0.0x00000.0xb2.0xEB",
"http://0x000c0.0x00000.0xb2.0xEB",
)
runTest(
"http://0xc0.0x00.0xb2.0xEB/bobo",
"http://0xc0.0x00.0xb2.0xEB/bobo",
)
runTest(
"ooh look i can find it in text http://0xc0.0x00.0xb2.0xEB/bobo like this",
"http://0xc0.0x00.0xb2.0xEB/bobo",
)
runTest(
"noscheme look 0xc0.0x00.0xb2.0xEB/bobo",
"0xc0.0x00.0xb2.0xEB/bobo",
)
runTest(
"no scheme 0xc0.0x00.0xb2.0xEB or path",
"0xc0.0x00.0xb2.0xEB",
)
}
@Test
fun testDottedOctalIpAddress() {
runTest(
"http://0301.0250.0002.0353",
"http://0301.0250.0002.0353",
)
runTest(
"http://0301.0250.0002.0353/bobo",
"http://0301.0250.0002.0353/bobo",
)
runTest("http://192.168.017.015/", "http://192.168.017.015/")
runTest(
"ooh look i can find it in text http://0301.0250.0002.0353/bobo like this",
"http://0301.0250.0002.0353/bobo",
)
runTest(
"noscheme look 0301.0250.0002.0353/bobo",
"0301.0250.0002.0353/bobo",
)
runTest(
"no scheme 0301.0250.0002.0353 or path",
"0301.0250.0002.0353",
)
}
@Test
fun testHexIpAddress() {
runTest("http://0xC00002EB/hello", "http://0xC00002EB/hello")
runTest(
"http://0xC00002EB.com/hello",
"http://0xC00002EB.com/hello",
)
runTest(
"still look it up as a normal url http://0xC00002EXsB.com/hello",
"http://0xC00002EXsB.com/hello",
)
runTest(
"ooh look i can find it in text http://0xC00002EB/bobo like this",
"http://0xC00002EB/bobo",
)
runTest(
"browsers dont support this without a scheme look 0xC00002EB/bobo",
)
}
@Test
fun testOctalIpAddress() {
runTest(
"http://030000001353/bobobo",
"http://030000001353/bobobo",
)
runTest(
"ooh look i can find it in text http://030000001353/bobo like this",
"http://030000001353/bobo",
)
runTest(
"browsers dont support this without a scheme look 030000001353/bobo",
)
}
@Test
fun testUrlWithEmptyPort() {
runTest(
"http://wtfismyip.com://foo.html",
"http://wtfismyip.com://foo.html",
)
runTest(
"make sure its right here http://wtfismyip.com://foo.html",
"http://wtfismyip.com://foo.html",
)
}
@Test
fun testUrlEncodedDot() {
runTest("hello www%2ewtfismyip%2ecom", "www%2ewtfismyip%2ecom")
runTest("hello wtfismyip%2ecom", "wtfismyip%2ecom")
runTest("http://wtfismyip%2ecom", "http://wtfismyip%2ecom")
runTest(
"make sure its right here http://wtfismyip%2ecom",
"http://wtfismyip%2ecom",
)
}
@Test
fun testUrlEncodedBadPath() {
runTest("%2ewtfismyip")
runTest("wtfismyip%2e")
runTest("wtfismyip%2ecom%2e", "wtfismyip%2ecom%2e")
runTest("wtfismyip%2ecom.", "wtfismyip%2ecom.")
runTest("%2ewtfismyip%2ecom", "wtfismyip%2ecom")
}
@Test
fun testDetectUrlEncoded() {
runTest(
"%77%77%77%2e%67%75%6d%62%6c%61%72%2e%63%6e",
"%77%77%77%2e%67%75%6d%62%6c%61%72%2e%63%6e",
)
runTest(
" asdf %77%77%77%2e%67%75%6d%62%6c%61%72%2e%63%6e",
"%77%77%77%2e%67%75%6d%62%6c%61%72%2e%63%6e",
)
runTest(
"%77%77%77%2e%67%75%6d%62%6c%61%72%2e%63%6e%2e",
"%77%77%77%2e%67%75%6d%62%6c%61%72%2e%63%6e%2e",
)
}
@Test
fun testIncompleteIpAddresses() {
runTest("hello 10...")
runTest("hello 10...1")
runTest("hello 10..1.")
runTest("hello 10..1.1")
runTest("hello 10.1..1")
runTest("hello 10.1.1.")
runTest("hello .192..")
runTest("hello .192..1")
runTest("hello .192.1.")
runTest("hello .192.1.1")
runTest("hello ..3.")
runTest("hello ..3.1")
runTest("hello ...1")
}
@Test
fun testIPv4EncodedDot() {
runTest("hello 192%2e168%2e1%2e1", "192%2e168%2e1%2e1")
runTest(
"hello 192.168%2e1%2e1/lalala",
"192.168%2e1%2e1/lalala",
)
}
@Test
fun testIPv4HexEncodedDot() {
runTest(
"hello 0xee%2e0xbb%2e0x1%2e0x1",
"0xee%2e0xbb%2e0x1%2e0x1",
)
runTest(
"hello 0xee%2e0xbb.0x1%2e0x1/lalala",
"0xee%2e0xbb.0x1%2e0x1/lalala",
)
}
@Test
fun testIpv6BadWithGoodUrls() {
runTest("[:::] [::] [bacd::]", "[::]", "[bacd::]")
runTest("[:0][::]", "[::]")
runTest("[:0:][::afaf]", "[::afaf]")
runTest(
"::] [fe80:aaaa:aaaa:aaaa::]",
"[fe80:aaaa:aaaa:aaaa::]",
)
runTest(
"fe80:22:]3123:[adf] [fe80:aaaa:aaaa:aaaa::]",
"[fe80:aaaa:aaaa:aaaa::]",
)
runTest("[][123[][ae][fae][de][:a][d]aef:E][f")
runTest("[][][]2[d][]][]]]:d][[[:d[e][aee:]af:")
}
@Test
fun testIpv6BadWithGoodUrlsEmbedded() {
runTest(
"[fe80:aaaa:aaaa:aaaa:[::]3dd0:7f8e:57b7:34d5f]",
"[::]",
)
runTest("[b[::7f8e]:55]akjef[::]", "[::7f8e]:55", "[::]")
runTest(
"[bcad::kkkk:aaaa:3dd0[::7f8e]:57b7:34d5]akjef[::]",
"[::7f8e]:57",
"[::]",
)
}
@Test
fun testIpv6BadWithGoodUrlsWeirder() {
runTest("[:[::]", "[::]")
runTest("[:] [feed::]", "[feed::]")
runTest(":[::feee]:]", "[::feee]")
runTest(":[::feee]:]]", "[::feee]")
runTest("[[:[::feee]:]", "[::feee]")
}
@Test
fun testIpv6ConsecutiveGoodUrls() {
runTest("[::afaf][eaea::][::]", "[::afaf]", "[eaea::]", "[::]")
runTest("[::afaf]www.google.com", "[::afaf]", "www.google.com")
runTest("[lalala:we][::]", "[::]")
runTest("[::fe][::]", "[::fe]", "[::]")
runTest("[aaaa::][:0:][::afaf]", "[aaaa::]", "[::afaf]")
}
@Test
fun testIpv6BacktrackingUsernamePassword() {
runTest("check out my url:google.com", "google.com")
runTest(
"check out my url:[::BAD:DEAD:BEEF:2e80:0:0]",
"[::BAD:DEAD:BEEF:2e80:0:0]",
)
runTest(
"check out my url:[::BAD:DEAD:BEEF:2e80:0:0] ",
"[::BAD:DEAD:BEEF:2e80:0:0]",
)
}
@Test
fun testIpv6BacktrackingEmptyDomainName() {
runTest("check out my http:///[::2e80:0:0]", "[::2e80:0:0]")
runTest("check out my http://./[::2e80:0:0]", "[::2e80:0:0]")
}
@Test
fun testIpv6DoubleSchemeWithDomain() {
runTest("http://http://[::2e80:0:0]", "http://[::2e80:0:0]")
runTest(
"make sure its right here http://http://[::2e80:0:0]",
"http://[::2e80:0:0]",
)
}
@Test
fun testIpv6MultipleSchemes() {
runTest(
"http://http://http://[::2e80:0:0]",
"http://[::2e80:0:0]",
)
runTest(
"make sure its right here http://http://[::2e80:0:0]",
"http://[::2e80:0:0]",
)
runTest(
"http://ftp://https://[::2e80:0:0]",
"https://[::2e80:0:0]",
)
runTest(
"make sure its right here http://ftp://https://[::2e80:0:0]",
"https://[::2e80:0:0]",
)
}
@Test
fun testIpv6FtpWithUsernameAndPassword() {
runTest(
"ftp with username is ftp://username:password@[::2e80:0:0]",
"ftp://username:password@[::2e80:0:0]",
)
}
@Test
fun testIpv6NewLinesAndTabsAreDelimiters() {
runTest(
"Do newlines and tabs break? [::2e80:0:0]/hello/\nworld [::BEEF:ADD:BEEF]\t/stuff/ [AAbb:AAbb:AAbb::]/\thello [::2e80:0:0\u0000]/hello world",
"[::2e80:0:0]/hello/",
"[::BEEF:ADD:BEEF]",
"[AAbb:AAbb:AAbb::]/",
)
}
@Test
fun testIpv6WithPort() {
runTest(
"http://[AAbb:AAbb:AAbb::]:8080/helloworld",
"http://[AAbb:AAbb:AAbb::]:8080/helloworld",
)
}
@Test
fun testIpv6IncorrectParsingHtmlWithBadOptions() {
runTest(
"<a href=\"http://[::AAbb:]/\">google.com</a>",
"http://[::AAbb:]/\">google.com</a>",
)
}
@Test
fun testIpv6EmptyPort() {
runTest(
"http://[::AAbb:]://foo.html",
"http://[::AAbb:]://foo.html",
)
runTest(
"make sure its right here http://[::AAbb:]://foo.html",
"http://[::AAbb:]://foo.html",
)
}
@Test
fun testBacktrackInvalidUsernamePassword() {
runTest("http://hello:asdf.com", "asdf.com")
}
/*
* https://github.com/linkedin/URL-Detector/issues/12
*/
@Test
fun testIssue12() {
runTest(
"http://user:pass@host.com host.com",
"http://user:pass@host.com",
"host.com",
)
}
/*
* https://github.com/linkedin/URL-Detector/issues/15
*/
@Test
fun testIssue15() {
runTest(
".............:::::::::::;;;;;;;;;;;;;;;::...............................................:::::::::::::::::::::::::::::....................",
)
}
/*
* https://github.com/linkedin/URL-Detector/issues/16
*/
@Test
fun testIssue16() {
runTest("://VIVE MARINE LE PEN//:@.")
}
@Test
fun testColonWithoutSlashesFail() {
val parser = UrlDetector("ftp:example.com")
val found: List<Url> = parser.detect()
for (url in found) {
assertEquals(url.scheme, "https")
// Should be detected as a username now and set to default http://
assertEquals(url.host, "example.com")
}
}
@Test
fun testIssueUnderscore() {
runTest("Neomobius_at_mstdn.jp@mostr.pub", "Neomobius_at_mstdn.jp@mostr.pub")
}
private fun runTest(
text: String,
vararg expected: String?,
) = assertEquals(
expected.toList(),
UrlDetector(text).detect().map { it.originalUrl },
)
}
@@ -20,7 +20,6 @@
*/
package com.vitorpamplona.quartz.utils
import com.linkedin.urls.detection.UrlDetector
import com.linkedin.urls.detection.UrlDetectorOptions
import com.vitorpamplona.quartz.utils.urldetector.detection.UrlDetector
actual fun fastFindURLs(text: String): List<String> = UrlDetector(text, UrlDetectorOptions.Default).detect().map { it.originalUrl }
actual fun fastFindURLs(text: String): List<String> = UrlDetector(text).detect().map { it.originalUrl }
@@ -20,6 +20,9 @@
*/
package com.vitorpamplona.quartz.nip64Chess
import com.vitorpamplona.quartz.nip64Chess.jester.JesterEvent
import com.vitorpamplona.quartz.nip64Chess.jester.JesterGameEvents
import com.vitorpamplona.quartz.nip64Chess.jester.JesterProtocol
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertFalse