Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
168 changes: 160 additions & 8 deletions packages/shared/src/five08/resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -13,7 +13,7 @@
from datetime import datetime, timezone
from html import unescape
from typing import Any, cast
from urllib.parse import urljoin, urlsplit
from urllib.parse import urljoin, urlsplit, urlunsplit

from curl_cffi import CurlOpt, requests as curl_requests
from curl_cffi.requests import BrowserTypeLiteral, RequestsError
Expand DownExpand Up@@ -1286,9 +1286,10 @@ def _build_initial_external_source_candidates(
added_website_source_keys: set[str] = set()
website_budget = PROFILE_SOURCE_MAX_WEBSITES

explicit_website_links = self._coerce_website_links(explicit_personal_websites)
for website_url in explicit_website_links:
source_key = normalized_website_identity_key(website_url)
explicit_website_links = self._coerce_fetchable_website_links(
explicit_personal_websites
)
for website_url, source_key in explicit_website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand All@@ -1304,9 +1305,10 @@ def _build_initial_external_source_candidates(
)
)

website_links = self._coerce_website_links(contact.get("cWebsiteLink"))
for website_url in website_links:
source_key = normalized_website_identity_key(website_url)
website_links = self._coerce_fetchable_website_links(
contact.get("cWebsiteLink")
)
for website_url, source_key in website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand DownExpand Up@@ -1454,7 +1456,22 @@ def inspect_profile_source_fetch(
*,
allow_javascript_fallback: bool = False,
) -> ProfileSourceFetchDiagnostics:
response = self._fetch_external_profile_source_response(url)
response: ProfileSourceHttpResponse | None = None
last_fetch_error: ValueError | None = None
for candidate_url in self._iter_profile_source_fetch_urls(
url,
allow_javascript_fallback=allow_javascript_fallback,
):
try:
response = self._fetch_external_profile_source_response(candidate_url)
break
except ValueError as exc:
last_fetch_error = exc
if not self._is_retryable_profile_fetch_error(str(exc)):
raise

if response is None:
raise last_fetch_error or ValueError("Profile fetch failed")
extracted = ""
extraction_error: ValueError | None = None
try:
Expand DownExpand Up@@ -1513,6 +1530,82 @@ def inspect_profile_source_fetch(
error=error,
)

def _iter_profile_source_fetch_urls(
self,
url: str,
*,
allow_javascript_fallback: bool,
) -> list[str]:
candidates = [url]
if not allow_javascript_fallback:
return candidates

try:
parsed = urlsplit(url)
except Exception:
return candidates

if parsed.scheme.lower() != "https":
return candidates

host = (parsed.hostname or "").strip()
if not host:
return candidates

seen = {url}
alternate_netloc = self._alternate_profile_host_netloc(parsed)
if parsed.port in {None, 443} and alternate_netloc:
alternate_https_candidate = urlunsplit(
parsed._replace(netloc=alternate_netloc)
)
if alternate_https_candidate not in seen:
seen.add(alternate_https_candidate)
candidates.append(alternate_https_candidate)

return candidates

@staticmethod
def _is_retryable_profile_fetch_error(error: str) -> bool:
normalized = error.casefold()
return "profile fetch failed:" in normalized and any(
marker in normalized
for marker in (
"tls connect error",
"tlsv1_alert_internal_error",
"tlsv1 alert internal error",
"ssl:",
)
)

@staticmethod
def _alternate_profile_host_netloc(parsed: Any) -> str | None:
host = (parsed.hostname or "").strip()
if not host:
return None
if host.casefold().startswith("www."):
alternate_host = host[4:]
else:
alternate_host = f"www.{host}"
return ResumeProfileProcessor._swap_url_port(
parsed,
new_host=alternate_host,
new_port=parsed.port,
)

@staticmethod
def _swap_url_port(
parsed: Any,
*,
new_host: str | None = None,
new_port: int | None,
) -> str:
host = new_host or (parsed.hostname or "").strip()
if not host:
return parsed.netloc
if new_port is None:
return host
return f"{host}:{new_port}"

def _fetch_external_profile_source_response(
self, url: str
) -> ProfileSourceHttpResponse:
Expand DownExpand Up@@ -2274,6 +2367,37 @@ def _coerce_website_links(self, value: Any) -> list[str]:

return normalized

def _coerce_fetchable_website_links(self, value: Any) -> list[tuple[str, str]]:
if value is None:
return []

if isinstance(value, str):
raw_values = [item for item in value.split(",") if item.strip()]
elif isinstance(value, (list, tuple, set)):
raw_values = list(value)
else:
return []

normalized: list[tuple[str, str]] = []
seen: set[str] = set()
for raw_value in raw_values:
if not isinstance(raw_value, str):
continue
raw_candidate = raw_value.strip()
normalized_link = self._normalize_website_url(raw_candidate)
if normalized_link is None:
continue
fetchable_link = self._normalize_fetchable_website_url(raw_candidate)
if fetchable_link is None:
continue
dedupe_key = normalized_website_identity_key(normalized_link)
if dedupe_key is None or dedupe_key in seen:
continue
seen.add(dedupe_key)
normalized.append((fetchable_link, dedupe_key))

return normalized

def _merge_website_links(
self, *, existing: list[str], extracted: list[str]
) -> list[str]:
Expand DownExpand Up@@ -2310,6 +2434,34 @@ def _merge_website_links(
def _normalize_website_url(value: str) -> str | None:
return normalize_website_url(value, allow_scheme_less=True)

@staticmethod
def _normalize_fetchable_website_url(value: str) -> str | None:
candidate = value.strip().strip(")]},.;:")
if not candidate:
return None

lower_candidate = candidate.lower()
if lower_candidate.startswith("www."):
candidate = f"https://{candidate}"
elif not lower_candidate.startswith(("http://", "https://")):
if not re.match(
r"(?i)^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/?#].*)?$",
candidate,
):
return None
candidate = f"https://{candidate}"

try:
parsed = urlsplit(candidate)
except Exception:
return None

if "@" in parsed.netloc:
return None
if not (parsed.hostname or "").strip():
return None
return parsed.geturl().rstrip("/")

def _normalize_email_address(self, value: Any) -> str | None:
if not isinstance(value, str):
return None
Expand Down
142 changes: 141 additions & 1 deletion tests/unit/test_resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,7 +6,7 @@
from types import SimpleNamespace

import pytest
from unittest.mock import MagicMock, Mock, patch
from unittest.mock import MagicMock, Mock, call, patch

from curl_cffi import CurlOpt

Expand DownExpand Up@@ -691,6 +691,20 @@ def test_build_initial_external_source_candidates_caps_websites_globally() -> No
assert github_urls == ["https://github.com/octocat"]


def test_build_initial_external_source_candidates_preserves_www_host_from_crm() -> None:
"""CRM website candidates should keep the stored host for the first fetch."""
processor = ResumeProfileProcessor()

candidates = processor._build_initial_external_source_candidates(
contact={"cWebsiteLink": ["https://www.example.com/about"]},
)

assert len(candidates) == 1
assert candidates[0].label == "Personal Website"
assert candidates[0].url == "https://www.example.com/about"
assert candidates[0].source_key == "website:example.com/about"


def test_fetch_external_profile_sources_retries_alternate_candidate_after_failure() -> (
None
):
Expand DownExpand Up@@ -795,6 +809,132 @@ def test_fetch_external_profile_source_text_skips_browser_for_github_sources() -
)


def test_inspect_profile_source_fetch_retries_personal_site_tls_variants() -> None:
"""Personal website fetches should recover from apex TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://www.example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://www.example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://example.com/about"),
call("https://www.example.com/about"),
]


def test_inspect_profile_source_fetch_uses_stripped_host_as_www_fallback() -> None:
"""Stored www hosts should fall back to the apex host after TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://www.example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://www.example.com/about"),
call("https://example.com/about"),
]


def test_iter_profile_source_fetch_urls_keeps_https_only() -> None:
"""TLS retries should stay on HTTPS host variants without downgrading transport."""
processor = ResumeProfileProcessor()

assert processor._iter_profile_source_fetch_urls(
"https://example.com/about",
allow_javascript_fallback=True,
) == [
"https://example.com/about",
"https://www.example.com/about",
]
assert processor._iter_profile_source_fetch_urls(
"https://www.example.com/about",
allow_javascript_fallback=True,
) == [
"https://www.example.com/about",
"https://example.com/about",
]


def test_inspect_profile_source_fetch_does_not_retry_non_tls_errors() -> None:
"""Non-TLS fetch failures should keep the existing error behavior."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError("Profile fetch failed: 404 Client Error")
)

with pytest.raises(ValueError, match="404 Client Error"):
processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://example.com/about"
)


def test_inspect_profile_source_fetch_does_not_retry_github_tls_failures() -> None:
"""GitHub-style sources should not try personal-website recovery variants."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
)
)

with pytest.raises(ValueError, match="TLS connect error"):
processor.inspect_profile_source_fetch(
"https://github.com/octocat",
allow_javascript_fallback=False,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://github.com/octocat"
)


def test_validate_browser_profile_request_url_blocks_non_public_http_targets() -> None:
"""Browser fallback requests should reject non-public HTTP(S) targets."""
processor = ResumeProfileProcessor()
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
 blocks
(function() {
function addCopyButtons() {
document.querySelectorAll('pre code').forEach(function(codeBlock) {
if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
codeBlock.parentElement.setAttribute('data-copy-added', 'true');
var btn = document.createElement('button');
btn.textContent = 'Copy';
btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
btn.onmouseover = function() { this.style.opacity = '1'; };
btn.onmouseout = function() { this.style.opacity = '0.7'; };
btn.onclick = function() {
navigator.clipboard.writeText(codeBlock.textContent).then(function() {
btn.textContent = 'Copied!';
setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
});
};
codeBlock.parentElement.style.position = 'relative';
codeBlock.parentElement.appendChild(btn);
});
}
addCopyButtons();
// Re-run on dynamic content
var observer = new MutationObserver(addCopyButtons);
observer.observe(document.body, { childList: true, subtree: true });
})();
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
fix: preserve CRM website hosts during profile fetch by michaelmwu · Pull Request #228 · 508-dev/508-workflows · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
168 changes: 160 additions & 8 deletions packages/shared/src/five08/resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -13,7 +13,7 @@
from datetime import datetime, timezone
from html import unescape
from typing import Any, cast
from urllib.parse import urljoin, urlsplit
from urllib.parse import urljoin, urlsplit, urlunsplit

from curl_cffi import CurlOpt, requests as curl_requests
from curl_cffi.requests import BrowserTypeLiteral, RequestsError
Expand DownExpand Up@@ -1286,9 +1286,10 @@ def _build_initial_external_source_candidates(
added_website_source_keys: set[str] = set()
website_budget = PROFILE_SOURCE_MAX_WEBSITES

explicit_website_links = self._coerce_website_links(explicit_personal_websites)
for website_url in explicit_website_links:
source_key = normalized_website_identity_key(website_url)
explicit_website_links = self._coerce_fetchable_website_links(
explicit_personal_websites
)
for website_url, source_key in explicit_website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand All@@ -1304,9 +1305,10 @@ def _build_initial_external_source_candidates(
)
)

website_links = self._coerce_website_links(contact.get("cWebsiteLink"))
for website_url in website_links:
source_key = normalized_website_identity_key(website_url)
website_links = self._coerce_fetchable_website_links(
contact.get("cWebsiteLink")
)
for website_url, source_key in website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand DownExpand Up@@ -1454,7 +1456,22 @@ def inspect_profile_source_fetch(
*,
allow_javascript_fallback: bool = False,
) -> ProfileSourceFetchDiagnostics:
response = self._fetch_external_profile_source_response(url)
response: ProfileSourceHttpResponse | None = None
last_fetch_error: ValueError | None = None
for candidate_url in self._iter_profile_source_fetch_urls(
url,
allow_javascript_fallback=allow_javascript_fallback,
):
try:
response = self._fetch_external_profile_source_response(candidate_url)
break
except ValueError as exc:
last_fetch_error = exc
if not self._is_retryable_profile_fetch_error(str(exc)):
raise

if response is None:
raise last_fetch_error or ValueError("Profile fetch failed")
extracted = ""
extraction_error: ValueError | None = None
try:
Expand DownExpand Up@@ -1513,6 +1530,82 @@ def inspect_profile_source_fetch(
error=error,
)

def _iter_profile_source_fetch_urls(
self,
url: str,
*,
allow_javascript_fallback: bool,
) -> list[str]:
candidates = [url]
if not allow_javascript_fallback:
return candidates

try:
parsed = urlsplit(url)
except Exception:
return candidates

if parsed.scheme.lower() != "https":
return candidates

host = (parsed.hostname or "").strip()
if not host:
return candidates

seen = {url}
alternate_netloc = self._alternate_profile_host_netloc(parsed)
if parsed.port in {None, 443} and alternate_netloc:
alternate_https_candidate = urlunsplit(
parsed._replace(netloc=alternate_netloc)
)
if alternate_https_candidate not in seen:
seen.add(alternate_https_candidate)
candidates.append(alternate_https_candidate)

return candidates

@staticmethod
def _is_retryable_profile_fetch_error(error: str) -> bool:
normalized = error.casefold()
return "profile fetch failed:" in normalized and any(
marker in normalized
for marker in (
"tls connect error",
"tlsv1_alert_internal_error",
"tlsv1 alert internal error",
"ssl:",
)
)

@staticmethod
def _alternate_profile_host_netloc(parsed: Any) -> str | None:
host = (parsed.hostname or "").strip()
if not host:
return None
if host.casefold().startswith("www."):
alternate_host = host[4:]
else:
alternate_host = f"www.{host}"
return ResumeProfileProcessor._swap_url_port(
parsed,
new_host=alternate_host,
new_port=parsed.port,
)

@staticmethod
def _swap_url_port(
parsed: Any,
*,
new_host: str | None = None,
new_port: int | None,
) -> str:
host = new_host or (parsed.hostname or "").strip()
if not host:
return parsed.netloc
if new_port is None:
return host
return f"{host}:{new_port}"

def _fetch_external_profile_source_response(
self, url: str
) -> ProfileSourceHttpResponse:
Expand DownExpand Up@@ -2274,6 +2367,37 @@ def _coerce_website_links(self, value: Any) -> list[str]:

return normalized

def _coerce_fetchable_website_links(self, value: Any) -> list[tuple[str, str]]:
if value is None:
return []

if isinstance(value, str):
raw_values = [item for item in value.split(",") if item.strip()]
elif isinstance(value, (list, tuple, set)):
raw_values = list(value)
else:
return []

normalized: list[tuple[str, str]] = []
seen: set[str] = set()
for raw_value in raw_values:
if not isinstance(raw_value, str):
continue
raw_candidate = raw_value.strip()
normalized_link = self._normalize_website_url(raw_candidate)
if normalized_link is None:
continue
fetchable_link = self._normalize_fetchable_website_url(raw_candidate)
if fetchable_link is None:
continue
dedupe_key = normalized_website_identity_key(normalized_link)
if dedupe_key is None or dedupe_key in seen:
continue
seen.add(dedupe_key)
normalized.append((fetchable_link, dedupe_key))

return normalized

def _merge_website_links(
self, *, existing: list[str], extracted: list[str]
) -> list[str]:
Expand DownExpand Up@@ -2310,6 +2434,34 @@ def _merge_website_links(
def _normalize_website_url(value: str) -> str | None:
return normalize_website_url(value, allow_scheme_less=True)

@staticmethod
def _normalize_fetchable_website_url(value: str) -> str | None:
candidate = value.strip().strip(")]},.;:")
if not candidate:
return None

lower_candidate = candidate.lower()
if lower_candidate.startswith("www."):
candidate = f"https://{candidate}"
elif not lower_candidate.startswith(("http://", "https://")):
if not re.match(
r"(?i)^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/?#].*)?$",
candidate,
):
return None
candidate = f"https://{candidate}"

try:
parsed = urlsplit(candidate)
except Exception:
return None

if "@" in parsed.netloc:
return None
if not (parsed.hostname or "").strip():
return None
return parsed.geturl().rstrip("/")

def _normalize_email_address(self, value: Any) -> str | None:
if not isinstance(value, str):
return None
Expand Down
142 changes: 141 additions & 1 deletion tests/unit/test_resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,7 +6,7 @@
from types import SimpleNamespace

import pytest
from unittest.mock import MagicMock, Mock, patch
from unittest.mock import MagicMock, Mock, call, patch

from curl_cffi import CurlOpt

Expand DownExpand Up@@ -691,6 +691,20 @@ def test_build_initial_external_source_candidates_caps_websites_globally() -> No
assert github_urls == ["https://github.com/octocat"]


def test_build_initial_external_source_candidates_preserves_www_host_from_crm() -> None:
"""CRM website candidates should keep the stored host for the first fetch."""
processor = ResumeProfileProcessor()

candidates = processor._build_initial_external_source_candidates(
contact={"cWebsiteLink": ["https://www.example.com/about"]},
)

assert len(candidates) == 1
assert candidates[0].label == "Personal Website"
assert candidates[0].url == "https://www.example.com/about"
assert candidates[0].source_key == "website:example.com/about"


def test_fetch_external_profile_sources_retries_alternate_candidate_after_failure() -> (
None
):
Expand DownExpand Up@@ -795,6 +809,132 @@ def test_fetch_external_profile_source_text_skips_browser_for_github_sources() -
)


def test_inspect_profile_source_fetch_retries_personal_site_tls_variants() -> None:
"""Personal website fetches should recover from apex TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://www.example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://www.example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://example.com/about"),
call("https://www.example.com/about"),
]


def test_inspect_profile_source_fetch_uses_stripped_host_as_www_fallback() -> None:
"""Stored www hosts should fall back to the apex host after TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://www.example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://www.example.com/about"),
call("https://example.com/about"),
]


def test_iter_profile_source_fetch_urls_keeps_https_only() -> None:
"""TLS retries should stay on HTTPS host variants without downgrading transport."""
processor = ResumeProfileProcessor()

assert processor._iter_profile_source_fetch_urls(
"https://example.com/about",
allow_javascript_fallback=True,
) == [
"https://example.com/about",
"https://www.example.com/about",
]
assert processor._iter_profile_source_fetch_urls(
"https://www.example.com/about",
allow_javascript_fallback=True,
) == [
"https://www.example.com/about",
"https://example.com/about",
]


def test_inspect_profile_source_fetch_does_not_retry_non_tls_errors() -> None:
"""Non-TLS fetch failures should keep the existing error behavior."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError("Profile fetch failed: 404 Client Error")
)

with pytest.raises(ValueError, match="404 Client Error"):
processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://example.com/about"
)


def test_inspect_profile_source_fetch_does_not_retry_github_tls_failures() -> None:
"""GitHub-style sources should not try personal-website recovery variants."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
)
)

with pytest.raises(ValueError, match="TLS connect error"):
processor.inspect_profile_source_fetch(
"https://github.com/octocat",
allow_javascript_fallback=False,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://github.com/octocat"
)


def test_validate_browser_profile_request_url_blocks_non_public_http_targets() -> None:
"""Browser fallback requests should reject non-public HTTP(S) targets."""
processor = ResumeProfileProcessor()
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' fix: preserve CRM website hosts during profile fetch by michaelmwu · Pull Request #228 · 508-dev/508-workflows · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
168 changes: 160 additions & 8 deletions packages/shared/src/five08/resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -13,7 +13,7 @@
from datetime import datetime, timezone
from html import unescape
from typing import Any, cast
from urllib.parse import urljoin, urlsplit
from urllib.parse import urljoin, urlsplit, urlunsplit

from curl_cffi import CurlOpt, requests as curl_requests
from curl_cffi.requests import BrowserTypeLiteral, RequestsError
Expand DownExpand Up@@ -1286,9 +1286,10 @@ def _build_initial_external_source_candidates(
added_website_source_keys: set[str] = set()
website_budget = PROFILE_SOURCE_MAX_WEBSITES

explicit_website_links = self._coerce_website_links(explicit_personal_websites)
for website_url in explicit_website_links:
source_key = normalized_website_identity_key(website_url)
explicit_website_links = self._coerce_fetchable_website_links(
explicit_personal_websites
)
for website_url, source_key in explicit_website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand All@@ -1304,9 +1305,10 @@ def _build_initial_external_source_candidates(
)
)

website_links = self._coerce_website_links(contact.get("cWebsiteLink"))
for website_url in website_links:
source_key = normalized_website_identity_key(website_url)
website_links = self._coerce_fetchable_website_links(
contact.get("cWebsiteLink")
)
for website_url, source_key in website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand DownExpand Up@@ -1454,7 +1456,22 @@ def inspect_profile_source_fetch(
*,
allow_javascript_fallback: bool = False,
) -> ProfileSourceFetchDiagnostics:
response = self._fetch_external_profile_source_response(url)
response: ProfileSourceHttpResponse | None = None
last_fetch_error: ValueError | None = None
for candidate_url in self._iter_profile_source_fetch_urls(
url,
allow_javascript_fallback=allow_javascript_fallback,
):
try:
response = self._fetch_external_profile_source_response(candidate_url)
break
except ValueError as exc:
last_fetch_error = exc
if not self._is_retryable_profile_fetch_error(str(exc)):
raise

if response is None:
raise last_fetch_error or ValueError("Profile fetch failed")
extracted = ""
extraction_error: ValueError | None = None
try:
Expand DownExpand Up@@ -1513,6 +1530,82 @@ def inspect_profile_source_fetch(
error=error,
)

def _iter_profile_source_fetch_urls(
self,
url: str,
*,
allow_javascript_fallback: bool,
) -> list[str]:
candidates = [url]
if not allow_javascript_fallback:
return candidates

try:
parsed = urlsplit(url)
except Exception:
return candidates

if parsed.scheme.lower() != "https":
return candidates

host = (parsed.hostname or "").strip()
if not host:
return candidates

seen = {url}
alternate_netloc = self._alternate_profile_host_netloc(parsed)
if parsed.port in {None, 443} and alternate_netloc:
alternate_https_candidate = urlunsplit(
parsed._replace(netloc=alternate_netloc)
)
if alternate_https_candidate not in seen:
seen.add(alternate_https_candidate)
candidates.append(alternate_https_candidate)

return candidates

@staticmethod
def _is_retryable_profile_fetch_error(error: str) -> bool:
normalized = error.casefold()
return "profile fetch failed:" in normalized and any(
marker in normalized
for marker in (
"tls connect error",
"tlsv1_alert_internal_error",
"tlsv1 alert internal error",
"ssl:",
)
)

@staticmethod
def _alternate_profile_host_netloc(parsed: Any) -> str | None:
host = (parsed.hostname or "").strip()
if not host:
return None
if host.casefold().startswith("www."):
alternate_host = host[4:]
else:
alternate_host = f"www.{host}"
return ResumeProfileProcessor._swap_url_port(
parsed,
new_host=alternate_host,
new_port=parsed.port,
)

@staticmethod
def _swap_url_port(
parsed: Any,
*,
new_host: str | None = None,
new_port: int | None,
) -> str:
host = new_host or (parsed.hostname or "").strip()
if not host:
return parsed.netloc
if new_port is None:
return host
return f"{host}:{new_port}"

def _fetch_external_profile_source_response(
self, url: str
) -> ProfileSourceHttpResponse:
Expand DownExpand Up@@ -2274,6 +2367,37 @@ def _coerce_website_links(self, value: Any) -> list[str]:

return normalized

def _coerce_fetchable_website_links(self, value: Any) -> list[tuple[str, str]]:
if value is None:
return []

if isinstance(value, str):
raw_values = [item for item in value.split(",") if item.strip()]
elif isinstance(value, (list, tuple, set)):
raw_values = list(value)
else:
return []

normalized: list[tuple[str, str]] = []
seen: set[str] = set()
for raw_value in raw_values:
if not isinstance(raw_value, str):
continue
raw_candidate = raw_value.strip()
normalized_link = self._normalize_website_url(raw_candidate)
if normalized_link is None:
continue
fetchable_link = self._normalize_fetchable_website_url(raw_candidate)
if fetchable_link is None:
continue
dedupe_key = normalized_website_identity_key(normalized_link)
if dedupe_key is None or dedupe_key in seen:
continue
seen.add(dedupe_key)
normalized.append((fetchable_link, dedupe_key))

return normalized

def _merge_website_links(
self, *, existing: list[str], extracted: list[str]
) -> list[str]:
Expand DownExpand Up@@ -2310,6 +2434,34 @@ def _merge_website_links(
def _normalize_website_url(value: str) -> str | None:
return normalize_website_url(value, allow_scheme_less=True)

@staticmethod
def _normalize_fetchable_website_url(value: str) -> str | None:
candidate = value.strip().strip(")]},.;:")
if not candidate:
return None

lower_candidate = candidate.lower()
if lower_candidate.startswith("www."):
candidate = f"https://{candidate}"
elif not lower_candidate.startswith(("http://", "https://")):
if not re.match(
r"(?i)^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/?#].*)?$",
candidate,
):
return None
candidate = f"https://{candidate}"

try:
parsed = urlsplit(candidate)
except Exception:
return None

if "@" in parsed.netloc:
return None
if not (parsed.hostname or "").strip():
return None
return parsed.geturl().rstrip("/")

def _normalize_email_address(self, value: Any) -> str | None:
if not isinstance(value, str):
return None
Expand Down
142 changes: 141 additions & 1 deletion tests/unit/test_resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,7 +6,7 @@
from types import SimpleNamespace

import pytest
from unittest.mock import MagicMock, Mock, patch
from unittest.mock import MagicMock, Mock, call, patch

from curl_cffi import CurlOpt

Expand DownExpand Up@@ -691,6 +691,20 @@ def test_build_initial_external_source_candidates_caps_websites_globally() -> No
assert github_urls == ["https://github.com/octocat"]


def test_build_initial_external_source_candidates_preserves_www_host_from_crm() -> None:
"""CRM website candidates should keep the stored host for the first fetch."""
processor = ResumeProfileProcessor()

candidates = processor._build_initial_external_source_candidates(
contact={"cWebsiteLink": ["https://www.example.com/about"]},
)

assert len(candidates) == 1
assert candidates[0].label == "Personal Website"
assert candidates[0].url == "https://www.example.com/about"
assert candidates[0].source_key == "website:example.com/about"


def test_fetch_external_profile_sources_retries_alternate_candidate_after_failure() -> (
None
):
Expand DownExpand Up@@ -795,6 +809,132 @@ def test_fetch_external_profile_source_text_skips_browser_for_github_sources() -
)


def test_inspect_profile_source_fetch_retries_personal_site_tls_variants() -> None:
"""Personal website fetches should recover from apex TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://www.example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://www.example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://example.com/about"),
call("https://www.example.com/about"),
]


def test_inspect_profile_source_fetch_uses_stripped_host_as_www_fallback() -> None:
"""Stored www hosts should fall back to the apex host after TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://www.example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://www.example.com/about"),
call("https://example.com/about"),
]


def test_iter_profile_source_fetch_urls_keeps_https_only() -> None:
"""TLS retries should stay on HTTPS host variants without downgrading transport."""
processor = ResumeProfileProcessor()

assert processor._iter_profile_source_fetch_urls(
"https://example.com/about",
allow_javascript_fallback=True,
) == [
"https://example.com/about",
"https://www.example.com/about",
]
assert processor._iter_profile_source_fetch_urls(
"https://www.example.com/about",
allow_javascript_fallback=True,
) == [
"https://www.example.com/about",
"https://example.com/about",
]


def test_inspect_profile_source_fetch_does_not_retry_non_tls_errors() -> None:
"""Non-TLS fetch failures should keep the existing error behavior."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError("Profile fetch failed: 404 Client Error")
)

with pytest.raises(ValueError, match="404 Client Error"):
processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://example.com/about"
)


def test_inspect_profile_source_fetch_does_not_retry_github_tls_failures() -> None:
"""GitHub-style sources should not try personal-website recovery variants."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
)
)

with pytest.raises(ValueError, match="TLS connect error"):
processor.inspect_profile_source_fetch(
"https://github.com/octocat",
allow_javascript_fallback=False,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://github.com/octocat"
)


def test_validate_browser_profile_request_url_blocks_non_public_http_targets() -> None:
"""Browser fallback requests should reject non-public HTTP(S) targets."""
processor = ResumeProfileProcessor()
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' fix: preserve CRM website hosts during profile fetch by michaelmwu · Pull Request #228 · 508-dev/508-workflows · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
168 changes: 160 additions & 8 deletions packages/shared/src/five08/resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -13,7 +13,7 @@
from datetime import datetime, timezone
from html import unescape
from typing import Any, cast
from urllib.parse import urljoin, urlsplit
from urllib.parse import urljoin, urlsplit, urlunsplit

from curl_cffi import CurlOpt, requests as curl_requests
from curl_cffi.requests import BrowserTypeLiteral, RequestsError
Expand DownExpand Up@@ -1286,9 +1286,10 @@ def _build_initial_external_source_candidates(
added_website_source_keys: set[str] = set()
website_budget = PROFILE_SOURCE_MAX_WEBSITES

explicit_website_links = self._coerce_website_links(explicit_personal_websites)
for website_url in explicit_website_links:
source_key = normalized_website_identity_key(website_url)
explicit_website_links = self._coerce_fetchable_website_links(
explicit_personal_websites
)
for website_url, source_key in explicit_website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand All@@ -1304,9 +1305,10 @@ def _build_initial_external_source_candidates(
)
)

website_links = self._coerce_website_links(contact.get("cWebsiteLink"))
for website_url in website_links:
source_key = normalized_website_identity_key(website_url)
website_links = self._coerce_fetchable_website_links(
contact.get("cWebsiteLink")
)
for website_url, source_key in website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand DownExpand Up@@ -1454,7 +1456,22 @@ def inspect_profile_source_fetch(
*,
allow_javascript_fallback: bool = False,
) -> ProfileSourceFetchDiagnostics:
response = self._fetch_external_profile_source_response(url)
response: ProfileSourceHttpResponse | None = None
last_fetch_error: ValueError | None = None
for candidate_url in self._iter_profile_source_fetch_urls(
url,
allow_javascript_fallback=allow_javascript_fallback,
):
try:
response = self._fetch_external_profile_source_response(candidate_url)
break
except ValueError as exc:
last_fetch_error = exc
if not self._is_retryable_profile_fetch_error(str(exc)):
raise

if response is None:
raise last_fetch_error or ValueError("Profile fetch failed")
extracted = ""
extraction_error: ValueError | None = None
try:
Expand DownExpand Up@@ -1513,6 +1530,82 @@ def inspect_profile_source_fetch(
error=error,
)

def _iter_profile_source_fetch_urls(
self,
url: str,
*,
allow_javascript_fallback: bool,
) -> list[str]:
candidates = [url]
if not allow_javascript_fallback:
return candidates

try:
parsed = urlsplit(url)
except Exception:
return candidates

if parsed.scheme.lower() != "https":
return candidates

host = (parsed.hostname or "").strip()
if not host:
return candidates

seen = {url}
alternate_netloc = self._alternate_profile_host_netloc(parsed)
if parsed.port in {None, 443} and alternate_netloc:
alternate_https_candidate = urlunsplit(
parsed._replace(netloc=alternate_netloc)
)
if alternate_https_candidate not in seen:
seen.add(alternate_https_candidate)
candidates.append(alternate_https_candidate)

return candidates

@staticmethod
def _is_retryable_profile_fetch_error(error: str) -> bool:
normalized = error.casefold()
return "profile fetch failed:" in normalized and any(
marker in normalized
for marker in (
"tls connect error",
"tlsv1_alert_internal_error",
"tlsv1 alert internal error",
"ssl:",
)
)

@staticmethod
def _alternate_profile_host_netloc(parsed: Any) -> str | None:
host = (parsed.hostname or "").strip()
if not host:
return None
if host.casefold().startswith("www."):
alternate_host = host[4:]
else:
alternate_host = f"www.{host}"
return ResumeProfileProcessor._swap_url_port(
parsed,
new_host=alternate_host,
new_port=parsed.port,
)

@staticmethod
def _swap_url_port(
parsed: Any,
*,
new_host: str | None = None,
new_port: int | None,
) -> str:
host = new_host or (parsed.hostname or "").strip()
if not host:
return parsed.netloc
if new_port is None:
return host
return f"{host}:{new_port}"

def _fetch_external_profile_source_response(
self, url: str
) -> ProfileSourceHttpResponse:
Expand DownExpand Up@@ -2274,6 +2367,37 @@ def _coerce_website_links(self, value: Any) -> list[str]:

return normalized

def _coerce_fetchable_website_links(self, value: Any) -> list[tuple[str, str]]:
if value is None:
return []

if isinstance(value, str):
raw_values = [item for item in value.split(",") if item.strip()]
elif isinstance(value, (list, tuple, set)):
raw_values = list(value)
else:
return []

normalized: list[tuple[str, str]] = []
seen: set[str] = set()
for raw_value in raw_values:
if not isinstance(raw_value, str):
continue
raw_candidate = raw_value.strip()
normalized_link = self._normalize_website_url(raw_candidate)
if normalized_link is None:
continue
fetchable_link = self._normalize_fetchable_website_url(raw_candidate)
if fetchable_link is None:
continue
dedupe_key = normalized_website_identity_key(normalized_link)
if dedupe_key is None or dedupe_key in seen:
continue
seen.add(dedupe_key)
normalized.append((fetchable_link, dedupe_key))

return normalized

def _merge_website_links(
self, *, existing: list[str], extracted: list[str]
) -> list[str]:
Expand DownExpand Up@@ -2310,6 +2434,34 @@ def _merge_website_links(
def _normalize_website_url(value: str) -> str | None:
return normalize_website_url(value, allow_scheme_less=True)

@staticmethod
def _normalize_fetchable_website_url(value: str) -> str | None:
candidate = value.strip().strip(")]},.;:")
if not candidate:
return None

lower_candidate = candidate.lower()
if lower_candidate.startswith("www."):
candidate = f"https://{candidate}"
elif not lower_candidate.startswith(("http://", "https://")):
if not re.match(
r"(?i)^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/?#].*)?$",
candidate,
):
return None
candidate = f"https://{candidate}"

try:
parsed = urlsplit(candidate)
except Exception:
return None

if "@" in parsed.netloc:
return None
if not (parsed.hostname or "").strip():
return None
return parsed.geturl().rstrip("/")

def _normalize_email_address(self, value: Any) -> str | None:
if not isinstance(value, str):
return None
Expand Down
142 changes: 141 additions & 1 deletion tests/unit/test_resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,7 +6,7 @@
from types import SimpleNamespace

import pytest
from unittest.mock import MagicMock, Mock, patch
from unittest.mock import MagicMock, Mock, call, patch

from curl_cffi import CurlOpt

Expand DownExpand Up@@ -691,6 +691,20 @@ def test_build_initial_external_source_candidates_caps_websites_globally() -> No
assert github_urls == ["https://github.com/octocat"]


def test_build_initial_external_source_candidates_preserves_www_host_from_crm() -> None:
"""CRM website candidates should keep the stored host for the first fetch."""
processor = ResumeProfileProcessor()

candidates = processor._build_initial_external_source_candidates(
contact={"cWebsiteLink": ["https://www.example.com/about"]},
)

assert len(candidates) == 1
assert candidates[0].label == "Personal Website"
assert candidates[0].url == "https://www.example.com/about"
assert candidates[0].source_key == "website:example.com/about"


def test_fetch_external_profile_sources_retries_alternate_candidate_after_failure() -> (
None
):
Expand DownExpand Up@@ -795,6 +809,132 @@ def test_fetch_external_profile_source_text_skips_browser_for_github_sources() -
)


def test_inspect_profile_source_fetch_retries_personal_site_tls_variants() -> None:
"""Personal website fetches should recover from apex TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://www.example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://www.example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://example.com/about"),
call("https://www.example.com/about"),
]


def test_inspect_profile_source_fetch_uses_stripped_host_as_www_fallback() -> None:
"""Stored www hosts should fall back to the apex host after TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://www.example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://www.example.com/about"),
call("https://example.com/about"),
]


def test_iter_profile_source_fetch_urls_keeps_https_only() -> None:
"""TLS retries should stay on HTTPS host variants without downgrading transport."""
processor = ResumeProfileProcessor()

assert processor._iter_profile_source_fetch_urls(
"https://example.com/about",
allow_javascript_fallback=True,
) == [
"https://example.com/about",
"https://www.example.com/about",
]
assert processor._iter_profile_source_fetch_urls(
"https://www.example.com/about",
allow_javascript_fallback=True,
) == [
"https://www.example.com/about",
"https://example.com/about",
]


def test_inspect_profile_source_fetch_does_not_retry_non_tls_errors() -> None:
"""Non-TLS fetch failures should keep the existing error behavior."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError("Profile fetch failed: 404 Client Error")
)

with pytest.raises(ValueError, match="404 Client Error"):
processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://example.com/about"
)


def test_inspect_profile_source_fetch_does_not_retry_github_tls_failures() -> None:
"""GitHub-style sources should not try personal-website recovery variants."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
)
)

with pytest.raises(ValueError, match="TLS connect error"):
processor.inspect_profile_source_fetch(
"https://github.com/octocat",
allow_javascript_fallback=False,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://github.com/octocat"
)


def test_validate_browser_profile_request_url_blocks_non_public_http_targets() -> None:
"""Browser fallback requests should reject non-public HTTP(S) targets."""
processor = ResumeProfileProcessor()
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ' fix: preserve CRM website hosts during profile fetch by michaelmwu · Pull Request #228 · 508-dev/508-workflows · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
168 changes: 160 additions & 8 deletions packages/shared/src/five08/resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -13,7 +13,7 @@
from datetime import datetime, timezone
from html import unescape
from typing import Any, cast
from urllib.parse import urljoin, urlsplit
from urllib.parse import urljoin, urlsplit, urlunsplit

from curl_cffi import CurlOpt, requests as curl_requests
from curl_cffi.requests import BrowserTypeLiteral, RequestsError
Expand DownExpand Up@@ -1286,9 +1286,10 @@ def _build_initial_external_source_candidates(
added_website_source_keys: set[str] = set()
website_budget = PROFILE_SOURCE_MAX_WEBSITES

explicit_website_links = self._coerce_website_links(explicit_personal_websites)
for website_url in explicit_website_links:
source_key = normalized_website_identity_key(website_url)
explicit_website_links = self._coerce_fetchable_website_links(
explicit_personal_websites
)
for website_url, source_key in explicit_website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand All@@ -1304,9 +1305,10 @@ def _build_initial_external_source_candidates(
)
)

website_links = self._coerce_website_links(contact.get("cWebsiteLink"))
for website_url in website_links:
source_key = normalized_website_identity_key(website_url)
website_links = self._coerce_fetchable_website_links(
contact.get("cWebsiteLink")
)
for website_url, source_key in website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand DownExpand Up@@ -1454,7 +1456,22 @@ def inspect_profile_source_fetch(
*,
allow_javascript_fallback: bool = False,
) -> ProfileSourceFetchDiagnostics:
response = self._fetch_external_profile_source_response(url)
response: ProfileSourceHttpResponse | None = None
last_fetch_error: ValueError | None = None
for candidate_url in self._iter_profile_source_fetch_urls(
url,
allow_javascript_fallback=allow_javascript_fallback,
):
try:
response = self._fetch_external_profile_source_response(candidate_url)
break
except ValueError as exc:
last_fetch_error = exc
if not self._is_retryable_profile_fetch_error(str(exc)):
raise

if response is None:
raise last_fetch_error or ValueError("Profile fetch failed")
extracted = ""
extraction_error: ValueError | None = None
try:
Expand DownExpand Up@@ -1513,6 +1530,82 @@ def inspect_profile_source_fetch(
error=error,
)

def _iter_profile_source_fetch_urls(
self,
url: str,
*,
allow_javascript_fallback: bool,
) -> list[str]:
candidates = [url]
if not allow_javascript_fallback:
return candidates

try:
parsed = urlsplit(url)
except Exception:
return candidates

if parsed.scheme.lower() != "https":
return candidates

host = (parsed.hostname or "").strip()
if not host:
return candidates

seen = {url}
alternate_netloc = self._alternate_profile_host_netloc(parsed)
if parsed.port in {None, 443} and alternate_netloc:
alternate_https_candidate = urlunsplit(
parsed._replace(netloc=alternate_netloc)
)
if alternate_https_candidate not in seen:
seen.add(alternate_https_candidate)
candidates.append(alternate_https_candidate)

return candidates

@staticmethod
def _is_retryable_profile_fetch_error(error: str) -> bool:
normalized = error.casefold()
return "profile fetch failed:" in normalized and any(
marker in normalized
for marker in (
"tls connect error",
"tlsv1_alert_internal_error",
"tlsv1 alert internal error",
"ssl:",
)
)

@staticmethod
def _alternate_profile_host_netloc(parsed: Any) -> str | None:
host = (parsed.hostname or "").strip()
if not host:
return None
if host.casefold().startswith("www."):
alternate_host = host[4:]
else:
alternate_host = f"www.{host}"
return ResumeProfileProcessor._swap_url_port(
parsed,
new_host=alternate_host,
new_port=parsed.port,
)

@staticmethod
def _swap_url_port(
parsed: Any,
*,
new_host: str | None = None,
new_port: int | None,
) -> str:
host = new_host or (parsed.hostname or "").strip()
if not host:
return parsed.netloc
if new_port is None:
return host
return f"{host}:{new_port}"

def _fetch_external_profile_source_response(
self, url: str
) -> ProfileSourceHttpResponse:
Expand DownExpand Up@@ -2274,6 +2367,37 @@ def _coerce_website_links(self, value: Any) -> list[str]:

return normalized

def _coerce_fetchable_website_links(self, value: Any) -> list[tuple[str, str]]:
if value is None:
return []

if isinstance(value, str):
raw_values = [item for item in value.split(",") if item.strip()]
elif isinstance(value, (list, tuple, set)):
raw_values = list(value)
else:
return []

normalized: list[tuple[str, str]] = []
seen: set[str] = set()
for raw_value in raw_values:
if not isinstance(raw_value, str):
continue
raw_candidate = raw_value.strip()
normalized_link = self._normalize_website_url(raw_candidate)
if normalized_link is None:
continue
fetchable_link = self._normalize_fetchable_website_url(raw_candidate)
if fetchable_link is None:
continue
dedupe_key = normalized_website_identity_key(normalized_link)
if dedupe_key is None or dedupe_key in seen:
continue
seen.add(dedupe_key)
normalized.append((fetchable_link, dedupe_key))

return normalized

def _merge_website_links(
self, *, existing: list[str], extracted: list[str]
) -> list[str]:
Expand DownExpand Up@@ -2310,6 +2434,34 @@ def _merge_website_links(
def _normalize_website_url(value: str) -> str | None:
return normalize_website_url(value, allow_scheme_less=True)

@staticmethod
def _normalize_fetchable_website_url(value: str) -> str | None:
candidate = value.strip().strip(")]},.;:")
if not candidate:
return None

lower_candidate = candidate.lower()
if lower_candidate.startswith("www."):
candidate = f"https://{candidate}"
elif not lower_candidate.startswith(("http://", "https://")):
if not re.match(
r"(?i)^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/?#].*)?$",
candidate,
):
return None
candidate = f"https://{candidate}"

try:
parsed = urlsplit(candidate)
except Exception:
return None

if "@" in parsed.netloc:
return None
if not (parsed.hostname or "").strip():
return None
return parsed.geturl().rstrip("/")

def _normalize_email_address(self, value: Any) -> str | None:
if not isinstance(value, str):
return None
Expand Down
142 changes: 141 additions & 1 deletion tests/unit/test_resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,7 +6,7 @@
from types import SimpleNamespace

import pytest
from unittest.mock import MagicMock, Mock, patch
from unittest.mock import MagicMock, Mock, call, patch

from curl_cffi import CurlOpt

Expand DownExpand Up@@ -691,6 +691,20 @@ def test_build_initial_external_source_candidates_caps_websites_globally() -> No
assert github_urls == ["https://github.com/octocat"]


def test_build_initial_external_source_candidates_preserves_www_host_from_crm() -> None:
"""CRM website candidates should keep the stored host for the first fetch."""
processor = ResumeProfileProcessor()

candidates = processor._build_initial_external_source_candidates(
contact={"cWebsiteLink": ["https://www.example.com/about"]},
)

assert len(candidates) == 1
assert candidates[0].label == "Personal Website"
assert candidates[0].url == "https://www.example.com/about"
assert candidates[0].source_key == "website:example.com/about"


def test_fetch_external_profile_sources_retries_alternate_candidate_after_failure() -> (
None
):
Expand DownExpand Up@@ -795,6 +809,132 @@ def test_fetch_external_profile_source_text_skips_browser_for_github_sources() -
)


def test_inspect_profile_source_fetch_retries_personal_site_tls_variants() -> None:
"""Personal website fetches should recover from apex TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://www.example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://www.example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://example.com/about"),
call("https://www.example.com/about"),
]


def test_inspect_profile_source_fetch_uses_stripped_host_as_www_fallback() -> None:
"""Stored www hosts should fall back to the apex host after TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://www.example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://www.example.com/about"),
call("https://example.com/about"),
]


def test_iter_profile_source_fetch_urls_keeps_https_only() -> None:
"""TLS retries should stay on HTTPS host variants without downgrading transport."""
processor = ResumeProfileProcessor()

assert processor._iter_profile_source_fetch_urls(
"https://example.com/about",
allow_javascript_fallback=True,
) == [
"https://example.com/about",
"https://www.example.com/about",
]
assert processor._iter_profile_source_fetch_urls(
"https://www.example.com/about",
allow_javascript_fallback=True,
) == [
"https://www.example.com/about",
"https://example.com/about",
]


def test_inspect_profile_source_fetch_does_not_retry_non_tls_errors() -> None:
"""Non-TLS fetch failures should keep the existing error behavior."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError("Profile fetch failed: 404 Client Error")
)

with pytest.raises(ValueError, match="404 Client Error"):
processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://example.com/about"
)


def test_inspect_profile_source_fetch_does_not_retry_github_tls_failures() -> None:
"""GitHub-style sources should not try personal-website recovery variants."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
)
)

with pytest.raises(ValueError, match="TLS connect error"):
processor.inspect_profile_source_fetch(
"https://github.com/octocat",
allow_javascript_fallback=False,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://github.com/octocat"
)


def test_validate_browser_profile_request_url_blocks_non_public_http_targets() -> None:
"""Browser fallback requests should reject non-public HTTP(S) targets."""
processor = ResumeProfileProcessor()
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' fix: preserve CRM website hosts during profile fetch by michaelmwu · Pull Request #228 · 508-dev/508-workflows · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
168 changes: 160 additions & 8 deletions packages/shared/src/five08/resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -13,7 +13,7 @@
from datetime import datetime, timezone
from html import unescape
from typing import Any, cast
from urllib.parse import urljoin, urlsplit
from urllib.parse import urljoin, urlsplit, urlunsplit

from curl_cffi import CurlOpt, requests as curl_requests
from curl_cffi.requests import BrowserTypeLiteral, RequestsError
Expand DownExpand Up@@ -1286,9 +1286,10 @@ def _build_initial_external_source_candidates(
added_website_source_keys: set[str] = set()
website_budget = PROFILE_SOURCE_MAX_WEBSITES

explicit_website_links = self._coerce_website_links(explicit_personal_websites)
for website_url in explicit_website_links:
source_key = normalized_website_identity_key(website_url)
explicit_website_links = self._coerce_fetchable_website_links(
explicit_personal_websites
)
for website_url, source_key in explicit_website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand All@@ -1304,9 +1305,10 @@ def _build_initial_external_source_candidates(
)
)

website_links = self._coerce_website_links(contact.get("cWebsiteLink"))
for website_url in website_links:
source_key = normalized_website_identity_key(website_url)
website_links = self._coerce_fetchable_website_links(
contact.get("cWebsiteLink")
)
for website_url, source_key in website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand DownExpand Up@@ -1454,7 +1456,22 @@ def inspect_profile_source_fetch(
*,
allow_javascript_fallback: bool = False,
) -> ProfileSourceFetchDiagnostics:
response = self._fetch_external_profile_source_response(url)
response: ProfileSourceHttpResponse | None = None
last_fetch_error: ValueError | None = None
for candidate_url in self._iter_profile_source_fetch_urls(
url,
allow_javascript_fallback=allow_javascript_fallback,
):
try:
response = self._fetch_external_profile_source_response(candidate_url)
break
except ValueError as exc:
last_fetch_error = exc
if not self._is_retryable_profile_fetch_error(str(exc)):
raise

if response is None:
raise last_fetch_error or ValueError("Profile fetch failed")
extracted = ""
extraction_error: ValueError | None = None
try:
Expand DownExpand Up@@ -1513,6 +1530,82 @@ def inspect_profile_source_fetch(
error=error,
)

def _iter_profile_source_fetch_urls(
self,
url: str,
*,
allow_javascript_fallback: bool,
) -> list[str]:
candidates = [url]
if not allow_javascript_fallback:
return candidates

try:
parsed = urlsplit(url)
except Exception:
return candidates

if parsed.scheme.lower() != "https":
return candidates

host = (parsed.hostname or "").strip()
if not host:
return candidates

seen = {url}
alternate_netloc = self._alternate_profile_host_netloc(parsed)
if parsed.port in {None, 443} and alternate_netloc:
alternate_https_candidate = urlunsplit(
parsed._replace(netloc=alternate_netloc)
)
if alternate_https_candidate not in seen:
seen.add(alternate_https_candidate)
candidates.append(alternate_https_candidate)

return candidates

@staticmethod
def _is_retryable_profile_fetch_error(error: str) -> bool:
normalized = error.casefold()
return "profile fetch failed:" in normalized and any(
marker in normalized
for marker in (
"tls connect error",
"tlsv1_alert_internal_error",
"tlsv1 alert internal error",
"ssl:",
)
)

@staticmethod
def _alternate_profile_host_netloc(parsed: Any) -> str | None:
host = (parsed.hostname or "").strip()
if not host:
return None
if host.casefold().startswith("www."):
alternate_host = host[4:]
else:
alternate_host = f"www.{host}"
return ResumeProfileProcessor._swap_url_port(
parsed,
new_host=alternate_host,
new_port=parsed.port,
)

@staticmethod
def _swap_url_port(
parsed: Any,
*,
new_host: str | None = None,
new_port: int | None,
) -> str:
host = new_host or (parsed.hostname or "").strip()
if not host:
return parsed.netloc
if new_port is None:
return host
return f"{host}:{new_port}"

def _fetch_external_profile_source_response(
self, url: str
) -> ProfileSourceHttpResponse:
Expand DownExpand Up@@ -2274,6 +2367,37 @@ def _coerce_website_links(self, value: Any) -> list[str]:

return normalized

def _coerce_fetchable_website_links(self, value: Any) -> list[tuple[str, str]]:
if value is None:
return []

if isinstance(value, str):
raw_values = [item for item in value.split(",") if item.strip()]
elif isinstance(value, (list, tuple, set)):
raw_values = list(value)
else:
return []

normalized: list[tuple[str, str]] = []
seen: set[str] = set()
for raw_value in raw_values:
if not isinstance(raw_value, str):
continue
raw_candidate = raw_value.strip()
normalized_link = self._normalize_website_url(raw_candidate)
if normalized_link is None:
continue
fetchable_link = self._normalize_fetchable_website_url(raw_candidate)
if fetchable_link is None:
continue
dedupe_key = normalized_website_identity_key(normalized_link)
if dedupe_key is None or dedupe_key in seen:
continue
seen.add(dedupe_key)
normalized.append((fetchable_link, dedupe_key))

return normalized

def _merge_website_links(
self, *, existing: list[str], extracted: list[str]
) -> list[str]:
Expand DownExpand Up@@ -2310,6 +2434,34 @@ def _merge_website_links(
def _normalize_website_url(value: str) -> str | None:
return normalize_website_url(value, allow_scheme_less=True)

@staticmethod
def _normalize_fetchable_website_url(value: str) -> str | None:
candidate = value.strip().strip(")]},.;:")
if not candidate:
return None

lower_candidate = candidate.lower()
if lower_candidate.startswith("www."):
candidate = f"https://{candidate}"
elif not lower_candidate.startswith(("http://", "https://")):
if not re.match(
r"(?i)^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/?#].*)?$",
candidate,
):
return None
candidate = f"https://{candidate}"

try:
parsed = urlsplit(candidate)
except Exception:
return None

if "@" in parsed.netloc:
return None
if not (parsed.hostname or "").strip():
return None
return parsed.geturl().rstrip("/")

def _normalize_email_address(self, value: Any) -> str | None:
if not isinstance(value, str):
return None
Expand Down
142 changes: 141 additions & 1 deletion tests/unit/test_resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,7 +6,7 @@
from types import SimpleNamespace

import pytest
from unittest.mock import MagicMock, Mock, patch
from unittest.mock import MagicMock, Mock, call, patch

from curl_cffi import CurlOpt

Expand DownExpand Up@@ -691,6 +691,20 @@ def test_build_initial_external_source_candidates_caps_websites_globally() -> No
assert github_urls == ["https://github.com/octocat"]


def test_build_initial_external_source_candidates_preserves_www_host_from_crm() -> None:
"""CRM website candidates should keep the stored host for the first fetch."""
processor = ResumeProfileProcessor()

candidates = processor._build_initial_external_source_candidates(
contact={"cWebsiteLink": ["https://www.example.com/about"]},
)

assert len(candidates) == 1
assert candidates[0].label == "Personal Website"
assert candidates[0].url == "https://www.example.com/about"
assert candidates[0].source_key == "website:example.com/about"


def test_fetch_external_profile_sources_retries_alternate_candidate_after_failure() -> (
None
):
Expand DownExpand Up@@ -795,6 +809,132 @@ def test_fetch_external_profile_source_text_skips_browser_for_github_sources() -
)


def test_inspect_profile_source_fetch_retries_personal_site_tls_variants() -> None:
"""Personal website fetches should recover from apex TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://www.example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://www.example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://example.com/about"),
call("https://www.example.com/about"),
]


def test_inspect_profile_source_fetch_uses_stripped_host_as_www_fallback() -> None:
"""Stored www hosts should fall back to the apex host after TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://www.example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://www.example.com/about"),
call("https://example.com/about"),
]


def test_iter_profile_source_fetch_urls_keeps_https_only() -> None:
"""TLS retries should stay on HTTPS host variants without downgrading transport."""
processor = ResumeProfileProcessor()

assert processor._iter_profile_source_fetch_urls(
"https://example.com/about",
allow_javascript_fallback=True,
) == [
"https://example.com/about",
"https://www.example.com/about",
]
assert processor._iter_profile_source_fetch_urls(
"https://www.example.com/about",
allow_javascript_fallback=True,
) == [
"https://www.example.com/about",
"https://example.com/about",
]


def test_inspect_profile_source_fetch_does_not_retry_non_tls_errors() -> None:
"""Non-TLS fetch failures should keep the existing error behavior."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError("Profile fetch failed: 404 Client Error")
)

with pytest.raises(ValueError, match="404 Client Error"):
processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://example.com/about"
)


def test_inspect_profile_source_fetch_does_not_retry_github_tls_failures() -> None:
"""GitHub-style sources should not try personal-website recovery variants."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
)
)

with pytest.raises(ValueError, match="TLS connect error"):
processor.inspect_profile_source_fetch(
"https://github.com/octocat",
allow_javascript_fallback=False,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://github.com/octocat"
)


def test_validate_browser_profile_request_url_blocks_non_public_http_targets() -> None:
"""Browser fallback requests should reject non-public HTTP(S) targets."""
processor = ResumeProfileProcessor()
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' fix: preserve CRM website hosts during profile fetch by michaelmwu · Pull Request #228 · 508-dev/508-workflows · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
168 changes: 160 additions & 8 deletions packages/shared/src/five08/resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -13,7 +13,7 @@
from datetime import datetime, timezone
from html import unescape
from typing import Any, cast
from urllib.parse import urljoin, urlsplit
from urllib.parse import urljoin, urlsplit, urlunsplit

from curl_cffi import CurlOpt, requests as curl_requests
from curl_cffi.requests import BrowserTypeLiteral, RequestsError
Expand DownExpand Up@@ -1286,9 +1286,10 @@ def _build_initial_external_source_candidates(
added_website_source_keys: set[str] = set()
website_budget = PROFILE_SOURCE_MAX_WEBSITES

explicit_website_links = self._coerce_website_links(explicit_personal_websites)
for website_url in explicit_website_links:
source_key = normalized_website_identity_key(website_url)
explicit_website_links = self._coerce_fetchable_website_links(
explicit_personal_websites
)
for website_url, source_key in explicit_website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand All@@ -1304,9 +1305,10 @@ def _build_initial_external_source_candidates(
)
)

website_links = self._coerce_website_links(contact.get("cWebsiteLink"))
for website_url in website_links:
source_key = normalized_website_identity_key(website_url)
website_links = self._coerce_fetchable_website_links(
contact.get("cWebsiteLink")
)
for website_url, source_key in website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand DownExpand Up@@ -1454,7 +1456,22 @@ def inspect_profile_source_fetch(
*,
allow_javascript_fallback: bool = False,
) -> ProfileSourceFetchDiagnostics:
response = self._fetch_external_profile_source_response(url)
response: ProfileSourceHttpResponse | None = None
last_fetch_error: ValueError | None = None
for candidate_url in self._iter_profile_source_fetch_urls(
url,
allow_javascript_fallback=allow_javascript_fallback,
):
try:
response = self._fetch_external_profile_source_response(candidate_url)
break
except ValueError as exc:
last_fetch_error = exc
if not self._is_retryable_profile_fetch_error(str(exc)):
raise

if response is None:
raise last_fetch_error or ValueError("Profile fetch failed")
extracted = ""
extraction_error: ValueError | None = None
try:
Expand DownExpand Up@@ -1513,6 +1530,82 @@ def inspect_profile_source_fetch(
error=error,
)

def _iter_profile_source_fetch_urls(
self,
url: str,
*,
allow_javascript_fallback: bool,
) -> list[str]:
candidates = [url]
if not allow_javascript_fallback:
return candidates

try:
parsed = urlsplit(url)
except Exception:
return candidates

if parsed.scheme.lower() != "https":
return candidates

host = (parsed.hostname or "").strip()
if not host:
return candidates

seen = {url}
alternate_netloc = self._alternate_profile_host_netloc(parsed)
if parsed.port in {None, 443} and alternate_netloc:
alternate_https_candidate = urlunsplit(
parsed._replace(netloc=alternate_netloc)
)
if alternate_https_candidate not in seen:
seen.add(alternate_https_candidate)
candidates.append(alternate_https_candidate)

return candidates

@staticmethod
def _is_retryable_profile_fetch_error(error: str) -> bool:
normalized = error.casefold()
return "profile fetch failed:" in normalized and any(
marker in normalized
for marker in (
"tls connect error",
"tlsv1_alert_internal_error",
"tlsv1 alert internal error",
"ssl:",
)
)

@staticmethod
def _alternate_profile_host_netloc(parsed: Any) -> str | None:
host = (parsed.hostname or "").strip()
if not host:
return None
if host.casefold().startswith("www."):
alternate_host = host[4:]
else:
alternate_host = f"www.{host}"
return ResumeProfileProcessor._swap_url_port(
parsed,
new_host=alternate_host,
new_port=parsed.port,
)

@staticmethod
def _swap_url_port(
parsed: Any,
*,
new_host: str | None = None,
new_port: int | None,
) -> str:
host = new_host or (parsed.hostname or "").strip()
if not host:
return parsed.netloc
if new_port is None:
return host
return f"{host}:{new_port}"

def _fetch_external_profile_source_response(
self, url: str
) -> ProfileSourceHttpResponse:
Expand DownExpand Up@@ -2274,6 +2367,37 @@ def _coerce_website_links(self, value: Any) -> list[str]:

return normalized

def _coerce_fetchable_website_links(self, value: Any) -> list[tuple[str, str]]:
if value is None:
return []

if isinstance(value, str):
raw_values = [item for item in value.split(",") if item.strip()]
elif isinstance(value, (list, tuple, set)):
raw_values = list(value)
else:
return []

normalized: list[tuple[str, str]] = []
seen: set[str] = set()
for raw_value in raw_values:
if not isinstance(raw_value, str):
continue
raw_candidate = raw_value.strip()
normalized_link = self._normalize_website_url(raw_candidate)
if normalized_link is None:
continue
fetchable_link = self._normalize_fetchable_website_url(raw_candidate)
if fetchable_link is None:
continue
dedupe_key = normalized_website_identity_key(normalized_link)
if dedupe_key is None or dedupe_key in seen:
continue
seen.add(dedupe_key)
normalized.append((fetchable_link, dedupe_key))

return normalized

def _merge_website_links(
self, *, existing: list[str], extracted: list[str]
) -> list[str]:
Expand DownExpand Up@@ -2310,6 +2434,34 @@ def _merge_website_links(
def _normalize_website_url(value: str) -> str | None:
return normalize_website_url(value, allow_scheme_less=True)

@staticmethod
def _normalize_fetchable_website_url(value: str) -> str | None:
candidate = value.strip().strip(")]},.;:")
if not candidate:
return None

lower_candidate = candidate.lower()
if lower_candidate.startswith("www."):
candidate = f"https://{candidate}"
elif not lower_candidate.startswith(("http://", "https://")):
if not re.match(
r"(?i)^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/?#].*)?$",
candidate,
):
return None
candidate = f"https://{candidate}"

try:
parsed = urlsplit(candidate)
except Exception:
return None

if "@" in parsed.netloc:
return None
if not (parsed.hostname or "").strip():
return None
return parsed.geturl().rstrip("/")

def _normalize_email_address(self, value: Any) -> str | None:
if not isinstance(value, str):
return None
Expand Down
142 changes: 141 additions & 1 deletion tests/unit/test_resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,7 +6,7 @@
from types import SimpleNamespace

import pytest
from unittest.mock import MagicMock, Mock, patch
from unittest.mock import MagicMock, Mock, call, patch

from curl_cffi import CurlOpt

Expand DownExpand Up@@ -691,6 +691,20 @@ def test_build_initial_external_source_candidates_caps_websites_globally() -> No
assert github_urls == ["https://github.com/octocat"]


def test_build_initial_external_source_candidates_preserves_www_host_from_crm() -> None:
"""CRM website candidates should keep the stored host for the first fetch."""
processor = ResumeProfileProcessor()

candidates = processor._build_initial_external_source_candidates(
contact={"cWebsiteLink": ["https://www.example.com/about"]},
)

assert len(candidates) == 1
assert candidates[0].label == "Personal Website"
assert candidates[0].url == "https://www.example.com/about"
assert candidates[0].source_key == "website:example.com/about"


def test_fetch_external_profile_sources_retries_alternate_candidate_after_failure() -> (
None
):
Expand DownExpand Up@@ -795,6 +809,132 @@ def test_fetch_external_profile_source_text_skips_browser_for_github_sources() -
)


def test_inspect_profile_source_fetch_retries_personal_site_tls_variants() -> None:
"""Personal website fetches should recover from apex TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://www.example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://www.example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://example.com/about"),
call("https://www.example.com/about"),
]


def test_inspect_profile_source_fetch_uses_stripped_host_as_www_fallback() -> None:
"""Stored www hosts should fall back to the apex host after TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://www.example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://www.example.com/about"),
call("https://example.com/about"),
]


def test_iter_profile_source_fetch_urls_keeps_https_only() -> None:
"""TLS retries should stay on HTTPS host variants without downgrading transport."""
processor = ResumeProfileProcessor()

assert processor._iter_profile_source_fetch_urls(
"https://example.com/about",
allow_javascript_fallback=True,
) == [
"https://example.com/about",
"https://www.example.com/about",
]
assert processor._iter_profile_source_fetch_urls(
"https://www.example.com/about",
allow_javascript_fallback=True,
) == [
"https://www.example.com/about",
"https://example.com/about",
]


def test_inspect_profile_source_fetch_does_not_retry_non_tls_errors() -> None:
"""Non-TLS fetch failures should keep the existing error behavior."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError("Profile fetch failed: 404 Client Error")
)

with pytest.raises(ValueError, match="404 Client Error"):
processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://example.com/about"
)


def test_inspect_profile_source_fetch_does_not_retry_github_tls_failures() -> None:
"""GitHub-style sources should not try personal-website recovery variants."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
)
)

with pytest.raises(ValueError, match="TLS connect error"):
processor.inspect_profile_source_fetch(
"https://github.com/octocat",
allow_javascript_fallback=False,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://github.com/octocat"
)


def test_validate_browser_profile_request_url_blocks_non_public_http_targets() -> None:
"""Browser fallback requests should reject non-public HTTP(S) targets."""
processor = ResumeProfileProcessor()
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Universal Dark Mode - works on any site (function() { var enabled = true; function applyDarkMode() { if (!enabled) return; // Create style element if it doesn't exist var style = document.getElementById('universal-dark-mode-style'); if (!style) { style = document.createElement('style'); style.id = 'universal-dark-mode-style'; document.head.appendChild(style); } // Dark mode CSS - inverts colors but preserves images/video style.textContent = ' /* Invert everything except media */ html { filter: invert(1) hue-rotate(180deg) !important; background: #1a1a2e !important; } /* Restore images, videos, iframes, canvas */ img, video, iframe, canvas, svg, picture, [style*="background-image"] { filter: invert(1) hue-rotate(180deg) !important; } /* Preserve specific elements that should not be inverted */ .no-dark-mode, .no-dark-mode *, [data-theme="light"], [data-theme="light"], .ace_editor, .ace_editor *, .CodeMirror, .CodeMirror *, .monaco-editor, .monaco-editor *, .markdown-body pre, .markdown-body pre *, .highlight, .highlight *, pre code, pre code * { filter: none !important; } /* Fix common UI elements */ .modal, .popup, .dropdown-menu, .tooltip, .popover { filter: invert(1) hue-rotate(180deg) !important; background: #2d2d44 !important; border-color: #444 !important; } /* Scrollbars */ ::-webkit-scrollbar { background: #1a1a2e !important; } ::-webkit-scrollbar-thumb { background: #444 !important; } ::-webkit-scrollbar-thumb:hover { background: #555 !important; } /* Selection */ ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; } ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; } '; } function removeDarkMode() { var style = document.getElementById('universal-dark-mode-style'); if (style) style.remove(); } // Toggle with Alt+Shift+D document.addEventListener('keydown', function(e) { if (e.altKey && e.shiftKey && e.key === 'D') { e.preventDefault(); enabled = !enabled; if (enabled) { applyDarkMode(); console.log('[Universal Dark Mode] Enabled'); } else { removeDarkMode(); console.log('[Universal Dark Mode] Disabled'); } } }); // Apply on load applyDarkMode(); // Re-apply on dynamic content var observer = new MutationObserver(function(mutations) { if (enabled && !document.getElementById('universal-dark-mode-style')) { applyDarkMode(); } }); observer.observe(document.head, { childList: true }); console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle'); })(); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })(); fix: preserve CRM website hosts during profile fetch by michaelmwu · Pull Request #228 · 508-dev/508-workflows · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
168 changes: 160 additions & 8 deletions packages/shared/src/five08/resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -13,7 +13,7 @@
from datetime import datetime, timezone
from html import unescape
from typing import Any, cast
from urllib.parse import urljoin, urlsplit
from urllib.parse import urljoin, urlsplit, urlunsplit

from curl_cffi import CurlOpt, requests as curl_requests
from curl_cffi.requests import BrowserTypeLiteral, RequestsError
Expand DownExpand Up@@ -1286,9 +1286,10 @@ def _build_initial_external_source_candidates(
added_website_source_keys: set[str] = set()
website_budget = PROFILE_SOURCE_MAX_WEBSITES

explicit_website_links = self._coerce_website_links(explicit_personal_websites)
for website_url in explicit_website_links:
source_key = normalized_website_identity_key(website_url)
explicit_website_links = self._coerce_fetchable_website_links(
explicit_personal_websites
)
for website_url, source_key in explicit_website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand All@@ -1304,9 +1305,10 @@ def _build_initial_external_source_candidates(
)
)

website_links = self._coerce_website_links(contact.get("cWebsiteLink"))
for website_url in website_links:
source_key = normalized_website_identity_key(website_url)
website_links = self._coerce_fetchable_website_links(
contact.get("cWebsiteLink")
)
for website_url, source_key in website_links:
if not source_key or source_key in added_website_source_keys:
continue
if website_budget <= 0:
Expand DownExpand Up@@ -1454,7 +1456,22 @@ def inspect_profile_source_fetch(
*,
allow_javascript_fallback: bool = False,
) -> ProfileSourceFetchDiagnostics:
response = self._fetch_external_profile_source_response(url)
response: ProfileSourceHttpResponse | None = None
last_fetch_error: ValueError | None = None
for candidate_url in self._iter_profile_source_fetch_urls(
url,
allow_javascript_fallback=allow_javascript_fallback,
):
try:
response = self._fetch_external_profile_source_response(candidate_url)
break
except ValueError as exc:
last_fetch_error = exc
if not self._is_retryable_profile_fetch_error(str(exc)):
raise

if response is None:
raise last_fetch_error or ValueError("Profile fetch failed")
extracted = ""
extraction_error: ValueError | None = None
try:
Expand DownExpand Up@@ -1513,6 +1530,82 @@ def inspect_profile_source_fetch(
error=error,
)

def _iter_profile_source_fetch_urls(
self,
url: str,
*,
allow_javascript_fallback: bool,
) -> list[str]:
candidates = [url]
if not allow_javascript_fallback:
return candidates

try:
parsed = urlsplit(url)
except Exception:
return candidates

if parsed.scheme.lower() != "https":
return candidates

host = (parsed.hostname or "").strip()
if not host:
return candidates

seen = {url}
alternate_netloc = self._alternate_profile_host_netloc(parsed)
if parsed.port in {None, 443} and alternate_netloc:
alternate_https_candidate = urlunsplit(
parsed._replace(netloc=alternate_netloc)
)
if alternate_https_candidate not in seen:
seen.add(alternate_https_candidate)
candidates.append(alternate_https_candidate)

return candidates

@staticmethod
def _is_retryable_profile_fetch_error(error: str) -> bool:
normalized = error.casefold()
return "profile fetch failed:" in normalized and any(
marker in normalized
for marker in (
"tls connect error",
"tlsv1_alert_internal_error",
"tlsv1 alert internal error",
"ssl:",
)
)

@staticmethod
def _alternate_profile_host_netloc(parsed: Any) -> str | None:
host = (parsed.hostname or "").strip()
if not host:
return None
if host.casefold().startswith("www."):
alternate_host = host[4:]
else:
alternate_host = f"www.{host}"
return ResumeProfileProcessor._swap_url_port(
parsed,
new_host=alternate_host,
new_port=parsed.port,
)

@staticmethod
def _swap_url_port(
parsed: Any,
*,
new_host: str | None = None,
new_port: int | None,
) -> str:
host = new_host or (parsed.hostname or "").strip()
if not host:
return parsed.netloc
if new_port is None:
return host
return f"{host}:{new_port}"

def _fetch_external_profile_source_response(
self, url: str
) -> ProfileSourceHttpResponse:
Expand DownExpand Up@@ -2274,6 +2367,37 @@ def _coerce_website_links(self, value: Any) -> list[str]:

return normalized

def _coerce_fetchable_website_links(self, value: Any) -> list[tuple[str, str]]:
if value is None:
return []

if isinstance(value, str):
raw_values = [item for item in value.split(",") if item.strip()]
elif isinstance(value, (list, tuple, set)):
raw_values = list(value)
else:
return []

normalized: list[tuple[str, str]] = []
seen: set[str] = set()
for raw_value in raw_values:
if not isinstance(raw_value, str):
continue
raw_candidate = raw_value.strip()
normalized_link = self._normalize_website_url(raw_candidate)
if normalized_link is None:
continue
fetchable_link = self._normalize_fetchable_website_url(raw_candidate)
if fetchable_link is None:
continue
dedupe_key = normalized_website_identity_key(normalized_link)
if dedupe_key is None or dedupe_key in seen:
continue
seen.add(dedupe_key)
normalized.append((fetchable_link, dedupe_key))

return normalized

def _merge_website_links(
self, *, existing: list[str], extracted: list[str]
) -> list[str]:
Expand DownExpand Up@@ -2310,6 +2434,34 @@ def _merge_website_links(
def _normalize_website_url(value: str) -> str | None:
return normalize_website_url(value, allow_scheme_less=True)

@staticmethod
def _normalize_fetchable_website_url(value: str) -> str | None:
candidate = value.strip().strip(")]},.;:")
if not candidate:
return None

lower_candidate = candidate.lower()
if lower_candidate.startswith("www."):
candidate = f"https://{candidate}"
elif not lower_candidate.startswith(("http://", "https://")):
if not re.match(
r"(?i)^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/?#].*)?$",
candidate,
):
return None
candidate = f"https://{candidate}"

try:
parsed = urlsplit(candidate)
except Exception:
return None

if "@" in parsed.netloc:
return None
if not (parsed.hostname or "").strip():
return None
return parsed.geturl().rstrip("/")

def _normalize_email_address(self, value: Any) -> str | None:
if not isinstance(value, str):
return None
Expand Down
142 changes: 141 additions & 1 deletion tests/unit/test_resume_profile_processor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -6,7 +6,7 @@
from types import SimpleNamespace

import pytest
from unittest.mock import MagicMock, Mock, patch
from unittest.mock import MagicMock, Mock, call, patch

from curl_cffi import CurlOpt

Expand DownExpand Up@@ -691,6 +691,20 @@ def test_build_initial_external_source_candidates_caps_websites_globally() -> No
assert github_urls == ["https://github.com/octocat"]


def test_build_initial_external_source_candidates_preserves_www_host_from_crm() -> None:
"""CRM website candidates should keep the stored host for the first fetch."""
processor = ResumeProfileProcessor()

candidates = processor._build_initial_external_source_candidates(
contact={"cWebsiteLink": ["https://www.example.com/about"]},
)

assert len(candidates) == 1
assert candidates[0].label == "Personal Website"
assert candidates[0].url == "https://www.example.com/about"
assert candidates[0].source_key == "website:example.com/about"


def test_fetch_external_profile_sources_retries_alternate_candidate_after_failure() -> (
None
):
Expand DownExpand Up@@ -795,6 +809,132 @@ def test_fetch_external_profile_source_text_skips_browser_for_github_sources() -
)


def test_inspect_profile_source_fetch_retries_personal_site_tls_variants() -> None:
"""Personal website fetches should recover from apex TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://www.example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://www.example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://example.com/about"),
call("https://www.example.com/about"),
]


def test_inspect_profile_source_fetch_uses_stripped_host_as_www_fallback() -> None:
"""Stored www hosts should fall back to the apex host after TLS failures."""
processor = ResumeProfileProcessor()
http_response = ProfileSourceHttpResponse(
final_url="https://example.com/about",
status_code=200,
headers={"content-type": "text/html; charset=utf-8"},
body=b"<html><head><title>Portfolio</title></head><body>Builder</body></html>",
)
processor._fetch_external_profile_source_response = Mock(
side_effect=[
ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
),
http_response,
]
)

diagnostics = processor.inspect_profile_source_fetch(
"https://www.example.com/about",
allow_javascript_fallback=True,
)

assert diagnostics.final_url == "https://example.com/about"
assert diagnostics.selected_text == "Title: Portfolio\nPortfolio Builder"
assert processor._fetch_external_profile_source_response.call_args_list == [
call("https://www.example.com/about"),
call("https://example.com/about"),
]


def test_iter_profile_source_fetch_urls_keeps_https_only() -> None:
"""TLS retries should stay on HTTPS host variants without downgrading transport."""
processor = ResumeProfileProcessor()

assert processor._iter_profile_source_fetch_urls(
"https://example.com/about",
allow_javascript_fallback=True,
) == [
"https://example.com/about",
"https://www.example.com/about",
]
assert processor._iter_profile_source_fetch_urls(
"https://www.example.com/about",
allow_javascript_fallback=True,
) == [
"https://www.example.com/about",
"https://example.com/about",
]


def test_inspect_profile_source_fetch_does_not_retry_non_tls_errors() -> None:
"""Non-TLS fetch failures should keep the existing error behavior."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError("Profile fetch failed: 404 Client Error")
)

with pytest.raises(ValueError, match="404 Client Error"):
processor.inspect_profile_source_fetch(
"https://example.com/about",
allow_javascript_fallback=True,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://example.com/about"
)


def test_inspect_profile_source_fetch_does_not_retry_github_tls_failures() -> None:
"""GitHub-style sources should not try personal-website recovery variants."""
processor = ResumeProfileProcessor()
processor._fetch_external_profile_source_response = Mock(
side_effect=ValueError(
"Profile fetch failed: Failed to perform, curl: (35) TLS connect "
"error: error:10000438:SSL routines:OPENSSL_internal:"
"TLSV1_ALERT_INTERNAL_ERROR."
)
)

with pytest.raises(ValueError, match="TLS connect error"):
processor.inspect_profile_source_fetch(
"https://github.com/octocat",
allow_javascript_fallback=False,
)

processor._fetch_external_profile_source_response.assert_called_once_with(
"https://github.com/octocat"
)


def test_validate_browser_profile_request_url_blocks_non_public_http_targets() -> None:
"""Browser fallback requests should reject non-public HTTP(S) targets."""
processor = ResumeProfileProcessor()
Expand Down