Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/ingest/ap_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,12 +32,17 @@ def fetch_full_text(self, article_url):
try:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
context = browser.new_context(user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36",
viewport={"width": 1280, "height": 800})
page = context.new_page()

# browser = p.chromium.launch(headless=True)
# page = browser.new_page()

#change the article_url to the redirected one (or just the same)
article_url = self.resolve_google_news_redirect(article_url)
#print(article_url)
page.goto(article_url, wait_until="domcontentloaded", timeout=15000)
page.goto(article_url, wait_until="domcontentloaded", timeout=30000)

# Wait for the main article body to load
page.wait_for_selector('div.RichTextStoryBody', timeout=3000)
Expand All@@ -50,6 +55,6 @@ def fetch_full_text(self, article_url):
return full_text.strip()

except Exception as e:
#print(f"Playwright error fetching {article_url}: {e}")
print(f"Playwright error fetching {article_url}: {e}")
return ""

13 changes: 4 additions & 9 deletions src/ingest/base_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -46,14 +46,9 @@ def check_and_save_new_entries(self, using_celery=False):
)

for entry in feed.entries:
# print("\n--- ENTRY ---")
# for k, v in entry.items():
# print(f"{k}: {v}")
# print(entry.link)
formattedEntry = self.format_entry(entry)
save_entry(formattedEntry, using_celery)

def check_no_save_new_entries(self):
feed = feedparser.parse(self.RSS_URL)
all_entries = []

for entry in feed.entries:
all_entries.append(self.format_entry(entry))

return all_entries
14 changes: 12 additions & 2 deletions src/ingest/bbc_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,6 +10,16 @@ def fetch_full_text(self, article_url):
soup = BeautifulSoup(response.content, 'html.parser')

article = soup.find('article')

if not article:
return None

if article:
return(article.get_text())
# # Remove unwanted sections like "related content", "media", or "byline"
# for unwanted in article.select('[data-component="byline"], [data-component="media-block"], .bbc-1msyfg1, .bbc-1fxtbkn'): # classes may vary
# unwanted.decompose()

# Gather all paragraphs that are part of the article body
paragraphs = article.find_all('p')

cleaned_text = '\n\n'.join(p.get_text(strip=True) for p in paragraphs)
return cleaned_text
1 change: 1 addition & 0 deletions src/ingest/cnn_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

class CNNIngestor(BaseIngestor):
RSS_URL = "http://rss.cnn.com/rss/cnn_world.rss"
# This RSS feed is from 2023????????

def fetch_full_text(self, url):
try:
Expand Down
4 changes: 4 additions & 0 deletions src/ingest/save_to_database.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,6 +16,7 @@ def save_entry(entry, using_celery):

#dont save entries without body text
if entry["full_text"] == "" or entry["full_text"] == None:
print("Unable to fetch full text.")
return

#check if the entry has already been saved and if it has not then save it
Expand All@@ -34,10 +35,13 @@ def save_entry(entry, using_celery):
current_app.tasks[ner_task.name]
#TODO: The following line can be used when connected to EC2 to actually use a GPU
#ner_task.apply_async(args=[str(inserted_id)], queue='gpu')
print("Checkpoint 1")
ner_task.apply_async(args=[str(inserted_id)])
except NotRegistered:
# fallback to inline
print("Checkpoint 2")
ner_task(str(inserted_id))
else:
# Inline execution
print("Checkpoint 3")
ner_task(str(inserted_id))
16 changes: 10 additions & 6 deletions src/justinsight/celery.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,31 +16,35 @@
# "args": (),
# },

# NOT SAVING ANYTHING? WHAT HAPPENED
# Maybe we should just give up on getting AP to work...
# "check-APfeed-every-5-minutes": {
# "task": "justinsight.tasks.apLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# "check-BBCfeed-every-5-minutes": {
# "task": "justinsight.tasks.bbcLogger_task",
# "schedule": 5.0,
# "args": (),
# },
# Checked and good
"check-BBCfeed-every-5-minutes": {
"task": "justinsight.tasks.bbcLogger_task",
"schedule": 5.0,
"args": (),
},

# Checked and good
# "check-CBSfeed-every-5-minutes": {
# "task": "justinsight.tasks.cbsLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# Could not find a current RSS feed for CNN :(
# "check-CNNfeed-every-5-minutes": {
# "task": "justinsight.tasks.cnnLogger_task",
# "schedule": 5.0,
# "args": (),
# },

#I am here
# "check-LATIMESfeed-every-5-minutes": {
# "task": "justinsight.tasks.latimesLogger_task",
# "schedule": 5.0,
Expand Down
24 changes: 0 additions & 24 deletions src/justinsight/nlpthings.py

This file was deleted.

6 changes: 3 additions & 3 deletions src/justinsight/streamlitapp.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -28,9 +28,9 @@
columns_to_show = st.multiselect("Columns to display", options=df.columns.tolist(), default=df.columns.tolist())
st.dataframe(df[columns_to_show])#, use_container_width=True)

# for i, row in df[columns_to_show].iterrows():
# st.markdown(f"### Entry {i+1}")
# st.write(row.to_dict())
for i, row in df[columns_to_show].iterrows():
st.markdown(f"### Entry {i+1}")
st.write(row.to_dict())



10 changes: 0 additions & 10 deletions src/justinsight/tasks.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,10 +8,7 @@
from ingest.npr_ingestor import NPRIngestor
from ingest.nyt_ingestor import NYTIngestor
from ingest.usnews_ingestor import USNEWSIngestor
from .nlpthings import dummy_addToEntryInDB
from ingest.save_to_database import collection
from nlp.ner_core import NERCore
from bson import ObjectId

@shared_task
def sample_task():
Expand DownExpand Up@@ -80,13 +77,6 @@ def usnewsLogger_task():
ingestor.check_and_save_new_entries(using_celery=True) # this will invoke the inherited logic
return "USNEWS RSS Feed checked."


@shared_task
def runNER_task(entry_id):
print(f"New worker so we can use GPU on this entry id: {entry_id}")
dummy_addToEntryInDB(entry_id)
#Do GPU-dependent processing here

@shared_task
def ner_task(article_id):
# Process article with NER results
Expand Down
2 changes: 1 addition & 1 deletion src/nlp/ner_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,7 +4,7 @@

class NERCore(BaseCore):
def __init__(self):
print("constructing NER Core instance")
print(" NER Core instance constructing")
super().__init__(
task="ner",
model_name="dslim/bert-base-NER",
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
 blocks
(function() {
function addCopyButtons() {
document.querySelectorAll('pre code').forEach(function(codeBlock) {
if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
codeBlock.parentElement.setAttribute('data-copy-added', 'true');
var btn = document.createElement('button');
btn.textContent = 'Copy';
btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
btn.onmouseover = function() { this.style.opacity = '1'; };
btn.onmouseout = function() { this.style.opacity = '0.7'; };
btn.onclick = function() {
navigator.clipboard.writeText(codeBlock.textContent).then(function() {
btn.textContent = 'Copied!';
setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
});
};
codeBlock.parentElement.style.position = 'relative';
codeBlock.parentElement.appendChild(btn);
});
}
addCopyButtons();
// Re-run on dynamic content
var observer = new MutationObserver(addCopyButtons);
observer.observe(document.body, { childList: true, subtree: true });
})();
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Grace by graceannmad · Pull Request #44 · JustInternetAI/JustInsight · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/ingest/ap_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,12 +32,17 @@ def fetch_full_text(self, article_url):
try:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
context = browser.new_context(user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36",
viewport={"width": 1280, "height": 800})
page = context.new_page()

# browser = p.chromium.launch(headless=True)
# page = browser.new_page()

#change the article_url to the redirected one (or just the same)
article_url = self.resolve_google_news_redirect(article_url)
#print(article_url)
page.goto(article_url, wait_until="domcontentloaded", timeout=15000)
page.goto(article_url, wait_until="domcontentloaded", timeout=30000)

# Wait for the main article body to load
page.wait_for_selector('div.RichTextStoryBody', timeout=3000)
Expand All@@ -50,6 +55,6 @@ def fetch_full_text(self, article_url):
return full_text.strip()

except Exception as e:
#print(f"Playwright error fetching {article_url}: {e}")
print(f"Playwright error fetching {article_url}: {e}")
return ""

13 changes: 4 additions & 9 deletions src/ingest/base_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -46,14 +46,9 @@ def check_and_save_new_entries(self, using_celery=False):
)

for entry in feed.entries:
# print("\n--- ENTRY ---")
# for k, v in entry.items():
# print(f"{k}: {v}")
# print(entry.link)
formattedEntry = self.format_entry(entry)
save_entry(formattedEntry, using_celery)

def check_no_save_new_entries(self):
feed = feedparser.parse(self.RSS_URL)
all_entries = []

for entry in feed.entries:
all_entries.append(self.format_entry(entry))

return all_entries
14 changes: 12 additions & 2 deletions src/ingest/bbc_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,6 +10,16 @@ def fetch_full_text(self, article_url):
soup = BeautifulSoup(response.content, 'html.parser')

article = soup.find('article')

if not article:
return None

if article:
return(article.get_text())
# # Remove unwanted sections like "related content", "media", or "byline"
# for unwanted in article.select('[data-component="byline"], [data-component="media-block"], .bbc-1msyfg1, .bbc-1fxtbkn'): # classes may vary
# unwanted.decompose()

# Gather all paragraphs that are part of the article body
paragraphs = article.find_all('p')

cleaned_text = '\n\n'.join(p.get_text(strip=True) for p in paragraphs)
return cleaned_text
1 change: 1 addition & 0 deletions src/ingest/cnn_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

class CNNIngestor(BaseIngestor):
RSS_URL = "http://rss.cnn.com/rss/cnn_world.rss"
# This RSS feed is from 2023????????

def fetch_full_text(self, url):
try:
Expand Down
4 changes: 4 additions & 0 deletions src/ingest/save_to_database.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,6 +16,7 @@ def save_entry(entry, using_celery):

#dont save entries without body text
if entry["full_text"] == "" or entry["full_text"] == None:
print("Unable to fetch full text.")
return

#check if the entry has already been saved and if it has not then save it
Expand All@@ -34,10 +35,13 @@ def save_entry(entry, using_celery):
current_app.tasks[ner_task.name]
#TODO: The following line can be used when connected to EC2 to actually use a GPU
#ner_task.apply_async(args=[str(inserted_id)], queue='gpu')
print("Checkpoint 1")
ner_task.apply_async(args=[str(inserted_id)])
except NotRegistered:
# fallback to inline
print("Checkpoint 2")
ner_task(str(inserted_id))
else:
# Inline execution
print("Checkpoint 3")
ner_task(str(inserted_id))
16 changes: 10 additions & 6 deletions src/justinsight/celery.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,31 +16,35 @@
# "args": (),
# },

# NOT SAVING ANYTHING? WHAT HAPPENED
# Maybe we should just give up on getting AP to work...
# "check-APfeed-every-5-minutes": {
# "task": "justinsight.tasks.apLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# "check-BBCfeed-every-5-minutes": {
# "task": "justinsight.tasks.bbcLogger_task",
# "schedule": 5.0,
# "args": (),
# },
# Checked and good
"check-BBCfeed-every-5-minutes": {
"task": "justinsight.tasks.bbcLogger_task",
"schedule": 5.0,
"args": (),
},

# Checked and good
# "check-CBSfeed-every-5-minutes": {
# "task": "justinsight.tasks.cbsLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# Could not find a current RSS feed for CNN :(
# "check-CNNfeed-every-5-minutes": {
# "task": "justinsight.tasks.cnnLogger_task",
# "schedule": 5.0,
# "args": (),
# },

#I am here
# "check-LATIMESfeed-every-5-minutes": {
# "task": "justinsight.tasks.latimesLogger_task",
# "schedule": 5.0,
Expand Down
24 changes: 0 additions & 24 deletions src/justinsight/nlpthings.py

This file was deleted.

6 changes: 3 additions & 3 deletions src/justinsight/streamlitapp.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -28,9 +28,9 @@
columns_to_show = st.multiselect("Columns to display", options=df.columns.tolist(), default=df.columns.tolist())
st.dataframe(df[columns_to_show])#, use_container_width=True)

# for i, row in df[columns_to_show].iterrows():
# st.markdown(f"### Entry {i+1}")
# st.write(row.to_dict())
for i, row in df[columns_to_show].iterrows():
st.markdown(f"### Entry {i+1}")
st.write(row.to_dict())



10 changes: 0 additions & 10 deletions src/justinsight/tasks.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,10 +8,7 @@
from ingest.npr_ingestor import NPRIngestor
from ingest.nyt_ingestor import NYTIngestor
from ingest.usnews_ingestor import USNEWSIngestor
from .nlpthings import dummy_addToEntryInDB
from ingest.save_to_database import collection
from nlp.ner_core import NERCore
from bson import ObjectId

@shared_task
def sample_task():
Expand DownExpand Up@@ -80,13 +77,6 @@ def usnewsLogger_task():
ingestor.check_and_save_new_entries(using_celery=True) # this will invoke the inherited logic
return "USNEWS RSS Feed checked."


@shared_task
def runNER_task(entry_id):
print(f"New worker so we can use GPU on this entry id: {entry_id}")
dummy_addToEntryInDB(entry_id)
#Do GPU-dependent processing here

@shared_task
def ner_task(article_id):
# Process article with NER results
Expand Down
2 changes: 1 addition & 1 deletion src/nlp/ner_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,7 +4,7 @@

class NERCore(BaseCore):
def __init__(self):
print("constructing NER Core instance")
print(" NER Core instance constructing")
super().__init__(
task="ner",
model_name="dslim/bert-base-NER",
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Grace by graceannmad · Pull Request #44 · JustInternetAI/JustInsight · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/ingest/ap_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,12 +32,17 @@ def fetch_full_text(self, article_url):
try:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
context = browser.new_context(user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36",
viewport={"width": 1280, "height": 800})
page = context.new_page()

# browser = p.chromium.launch(headless=True)
# page = browser.new_page()

#change the article_url to the redirected one (or just the same)
article_url = self.resolve_google_news_redirect(article_url)
#print(article_url)
page.goto(article_url, wait_until="domcontentloaded", timeout=15000)
page.goto(article_url, wait_until="domcontentloaded", timeout=30000)

# Wait for the main article body to load
page.wait_for_selector('div.RichTextStoryBody', timeout=3000)
Expand All@@ -50,6 +55,6 @@ def fetch_full_text(self, article_url):
return full_text.strip()

except Exception as e:
#print(f"Playwright error fetching {article_url}: {e}")
print(f"Playwright error fetching {article_url}: {e}")
return ""

13 changes: 4 additions & 9 deletions src/ingest/base_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -46,14 +46,9 @@ def check_and_save_new_entries(self, using_celery=False):
)

for entry in feed.entries:
# print("\n--- ENTRY ---")
# for k, v in entry.items():
# print(f"{k}: {v}")
# print(entry.link)
formattedEntry = self.format_entry(entry)
save_entry(formattedEntry, using_celery)

def check_no_save_new_entries(self):
feed = feedparser.parse(self.RSS_URL)
all_entries = []

for entry in feed.entries:
all_entries.append(self.format_entry(entry))

return all_entries
14 changes: 12 additions & 2 deletions src/ingest/bbc_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,6 +10,16 @@ def fetch_full_text(self, article_url):
soup = BeautifulSoup(response.content, 'html.parser')

article = soup.find('article')

if not article:
return None

if article:
return(article.get_text())
# # Remove unwanted sections like "related content", "media", or "byline"
# for unwanted in article.select('[data-component="byline"], [data-component="media-block"], .bbc-1msyfg1, .bbc-1fxtbkn'): # classes may vary
# unwanted.decompose()

# Gather all paragraphs that are part of the article body
paragraphs = article.find_all('p')

cleaned_text = '\n\n'.join(p.get_text(strip=True) for p in paragraphs)
return cleaned_text
1 change: 1 addition & 0 deletions src/ingest/cnn_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

class CNNIngestor(BaseIngestor):
RSS_URL = "http://rss.cnn.com/rss/cnn_world.rss"
# This RSS feed is from 2023????????

def fetch_full_text(self, url):
try:
Expand Down
4 changes: 4 additions & 0 deletions src/ingest/save_to_database.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,6 +16,7 @@ def save_entry(entry, using_celery):

#dont save entries without body text
if entry["full_text"] == "" or entry["full_text"] == None:
print("Unable to fetch full text.")
return

#check if the entry has already been saved and if it has not then save it
Expand All@@ -34,10 +35,13 @@ def save_entry(entry, using_celery):
current_app.tasks[ner_task.name]
#TODO: The following line can be used when connected to EC2 to actually use a GPU
#ner_task.apply_async(args=[str(inserted_id)], queue='gpu')
print("Checkpoint 1")
ner_task.apply_async(args=[str(inserted_id)])
except NotRegistered:
# fallback to inline
print("Checkpoint 2")
ner_task(str(inserted_id))
else:
# Inline execution
print("Checkpoint 3")
ner_task(str(inserted_id))
16 changes: 10 additions & 6 deletions src/justinsight/celery.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,31 +16,35 @@
# "args": (),
# },

# NOT SAVING ANYTHING? WHAT HAPPENED
# Maybe we should just give up on getting AP to work...
# "check-APfeed-every-5-minutes": {
# "task": "justinsight.tasks.apLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# "check-BBCfeed-every-5-minutes": {
# "task": "justinsight.tasks.bbcLogger_task",
# "schedule": 5.0,
# "args": (),
# },
# Checked and good
"check-BBCfeed-every-5-minutes": {
"task": "justinsight.tasks.bbcLogger_task",
"schedule": 5.0,
"args": (),
},

# Checked and good
# "check-CBSfeed-every-5-minutes": {
# "task": "justinsight.tasks.cbsLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# Could not find a current RSS feed for CNN :(
# "check-CNNfeed-every-5-minutes": {
# "task": "justinsight.tasks.cnnLogger_task",
# "schedule": 5.0,
# "args": (),
# },

#I am here
# "check-LATIMESfeed-every-5-minutes": {
# "task": "justinsight.tasks.latimesLogger_task",
# "schedule": 5.0,
Expand Down
24 changes: 0 additions & 24 deletions src/justinsight/nlpthings.py

This file was deleted.

6 changes: 3 additions & 3 deletions src/justinsight/streamlitapp.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -28,9 +28,9 @@
columns_to_show = st.multiselect("Columns to display", options=df.columns.tolist(), default=df.columns.tolist())
st.dataframe(df[columns_to_show])#, use_container_width=True)

# for i, row in df[columns_to_show].iterrows():
# st.markdown(f"### Entry {i+1}")
# st.write(row.to_dict())
for i, row in df[columns_to_show].iterrows():
st.markdown(f"### Entry {i+1}")
st.write(row.to_dict())



10 changes: 0 additions & 10 deletions src/justinsight/tasks.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,10 +8,7 @@
from ingest.npr_ingestor import NPRIngestor
from ingest.nyt_ingestor import NYTIngestor
from ingest.usnews_ingestor import USNEWSIngestor
from .nlpthings import dummy_addToEntryInDB
from ingest.save_to_database import collection
from nlp.ner_core import NERCore
from bson import ObjectId

@shared_task
def sample_task():
Expand DownExpand Up@@ -80,13 +77,6 @@ def usnewsLogger_task():
ingestor.check_and_save_new_entries(using_celery=True) # this will invoke the inherited logic
return "USNEWS RSS Feed checked."


@shared_task
def runNER_task(entry_id):
print(f"New worker so we can use GPU on this entry id: {entry_id}")
dummy_addToEntryInDB(entry_id)
#Do GPU-dependent processing here

@shared_task
def ner_task(article_id):
# Process article with NER results
Expand Down
2 changes: 1 addition & 1 deletion src/nlp/ner_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,7 +4,7 @@

class NERCore(BaseCore):
def __init__(self):
print("constructing NER Core instance")
print(" NER Core instance constructing")
super().__init__(
task="ner",
model_name="dslim/bert-base-NER",
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Grace by graceannmad · Pull Request #44 · JustInternetAI/JustInsight · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/ingest/ap_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,12 +32,17 @@ def fetch_full_text(self, article_url):
try:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
context = browser.new_context(user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36",
viewport={"width": 1280, "height": 800})
page = context.new_page()

# browser = p.chromium.launch(headless=True)
# page = browser.new_page()

#change the article_url to the redirected one (or just the same)
article_url = self.resolve_google_news_redirect(article_url)
#print(article_url)
page.goto(article_url, wait_until="domcontentloaded", timeout=15000)
page.goto(article_url, wait_until="domcontentloaded", timeout=30000)

# Wait for the main article body to load
page.wait_for_selector('div.RichTextStoryBody', timeout=3000)
Expand All@@ -50,6 +55,6 @@ def fetch_full_text(self, article_url):
return full_text.strip()

except Exception as e:
#print(f"Playwright error fetching {article_url}: {e}")
print(f"Playwright error fetching {article_url}: {e}")
return ""

13 changes: 4 additions & 9 deletions src/ingest/base_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -46,14 +46,9 @@ def check_and_save_new_entries(self, using_celery=False):
)

for entry in feed.entries:
# print("\n--- ENTRY ---")
# for k, v in entry.items():
# print(f"{k}: {v}")
# print(entry.link)
formattedEntry = self.format_entry(entry)
save_entry(formattedEntry, using_celery)

def check_no_save_new_entries(self):
feed = feedparser.parse(self.RSS_URL)
all_entries = []

for entry in feed.entries:
all_entries.append(self.format_entry(entry))

return all_entries
14 changes: 12 additions & 2 deletions src/ingest/bbc_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,6 +10,16 @@ def fetch_full_text(self, article_url):
soup = BeautifulSoup(response.content, 'html.parser')

article = soup.find('article')

if not article:
return None

if article:
return(article.get_text())
# # Remove unwanted sections like "related content", "media", or "byline"
# for unwanted in article.select('[data-component="byline"], [data-component="media-block"], .bbc-1msyfg1, .bbc-1fxtbkn'): # classes may vary
# unwanted.decompose()

# Gather all paragraphs that are part of the article body
paragraphs = article.find_all('p')

cleaned_text = '\n\n'.join(p.get_text(strip=True) for p in paragraphs)
return cleaned_text
1 change: 1 addition & 0 deletions src/ingest/cnn_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

class CNNIngestor(BaseIngestor):
RSS_URL = "http://rss.cnn.com/rss/cnn_world.rss"
# This RSS feed is from 2023????????

def fetch_full_text(self, url):
try:
Expand Down
4 changes: 4 additions & 0 deletions src/ingest/save_to_database.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,6 +16,7 @@ def save_entry(entry, using_celery):

#dont save entries without body text
if entry["full_text"] == "" or entry["full_text"] == None:
print("Unable to fetch full text.")
return

#check if the entry has already been saved and if it has not then save it
Expand All@@ -34,10 +35,13 @@ def save_entry(entry, using_celery):
current_app.tasks[ner_task.name]
#TODO: The following line can be used when connected to EC2 to actually use a GPU
#ner_task.apply_async(args=[str(inserted_id)], queue='gpu')
print("Checkpoint 1")
ner_task.apply_async(args=[str(inserted_id)])
except NotRegistered:
# fallback to inline
print("Checkpoint 2")
ner_task(str(inserted_id))
else:
# Inline execution
print("Checkpoint 3")
ner_task(str(inserted_id))
16 changes: 10 additions & 6 deletions src/justinsight/celery.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,31 +16,35 @@
# "args": (),
# },

# NOT SAVING ANYTHING? WHAT HAPPENED
# Maybe we should just give up on getting AP to work...
# "check-APfeed-every-5-minutes": {
# "task": "justinsight.tasks.apLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# "check-BBCfeed-every-5-minutes": {
# "task": "justinsight.tasks.bbcLogger_task",
# "schedule": 5.0,
# "args": (),
# },
# Checked and good
"check-BBCfeed-every-5-minutes": {
"task": "justinsight.tasks.bbcLogger_task",
"schedule": 5.0,
"args": (),
},

# Checked and good
# "check-CBSfeed-every-5-minutes": {
# "task": "justinsight.tasks.cbsLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# Could not find a current RSS feed for CNN :(
# "check-CNNfeed-every-5-minutes": {
# "task": "justinsight.tasks.cnnLogger_task",
# "schedule": 5.0,
# "args": (),
# },

#I am here
# "check-LATIMESfeed-every-5-minutes": {
# "task": "justinsight.tasks.latimesLogger_task",
# "schedule": 5.0,
Expand Down
24 changes: 0 additions & 24 deletions src/justinsight/nlpthings.py

This file was deleted.

6 changes: 3 additions & 3 deletions src/justinsight/streamlitapp.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -28,9 +28,9 @@
columns_to_show = st.multiselect("Columns to display", options=df.columns.tolist(), default=df.columns.tolist())
st.dataframe(df[columns_to_show])#, use_container_width=True)

# for i, row in df[columns_to_show].iterrows():
# st.markdown(f"### Entry {i+1}")
# st.write(row.to_dict())
for i, row in df[columns_to_show].iterrows():
st.markdown(f"### Entry {i+1}")
st.write(row.to_dict())



10 changes: 0 additions & 10 deletions src/justinsight/tasks.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,10 +8,7 @@
from ingest.npr_ingestor import NPRIngestor
from ingest.nyt_ingestor import NYTIngestor
from ingest.usnews_ingestor import USNEWSIngestor
from .nlpthings import dummy_addToEntryInDB
from ingest.save_to_database import collection
from nlp.ner_core import NERCore
from bson import ObjectId

@shared_task
def sample_task():
Expand DownExpand Up@@ -80,13 +77,6 @@ def usnewsLogger_task():
ingestor.check_and_save_new_entries(using_celery=True) # this will invoke the inherited logic
return "USNEWS RSS Feed checked."


@shared_task
def runNER_task(entry_id):
print(f"New worker so we can use GPU on this entry id: {entry_id}")
dummy_addToEntryInDB(entry_id)
#Do GPU-dependent processing here

@shared_task
def ner_task(article_id):
# Process article with NER results
Expand Down
2 changes: 1 addition & 1 deletion src/nlp/ner_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,7 +4,7 @@

class NERCore(BaseCore):
def __init__(self):
print("constructing NER Core instance")
print(" NER Core instance constructing")
super().__init__(
task="ner",
model_name="dslim/bert-base-NER",
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ' Grace by graceannmad · Pull Request #44 · JustInternetAI/JustInsight · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/ingest/ap_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,12 +32,17 @@ def fetch_full_text(self, article_url):
try:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
context = browser.new_context(user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36",
viewport={"width": 1280, "height": 800})
page = context.new_page()

# browser = p.chromium.launch(headless=True)
# page = browser.new_page()

#change the article_url to the redirected one (or just the same)
article_url = self.resolve_google_news_redirect(article_url)
#print(article_url)
page.goto(article_url, wait_until="domcontentloaded", timeout=15000)
page.goto(article_url, wait_until="domcontentloaded", timeout=30000)

# Wait for the main article body to load
page.wait_for_selector('div.RichTextStoryBody', timeout=3000)
Expand All@@ -50,6 +55,6 @@ def fetch_full_text(self, article_url):
return full_text.strip()

except Exception as e:
#print(f"Playwright error fetching {article_url}: {e}")
print(f"Playwright error fetching {article_url}: {e}")
return ""

13 changes: 4 additions & 9 deletions src/ingest/base_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -46,14 +46,9 @@ def check_and_save_new_entries(self, using_celery=False):
)

for entry in feed.entries:
# print("\n--- ENTRY ---")
# for k, v in entry.items():
# print(f"{k}: {v}")
# print(entry.link)
formattedEntry = self.format_entry(entry)
save_entry(formattedEntry, using_celery)

def check_no_save_new_entries(self):
feed = feedparser.parse(self.RSS_URL)
all_entries = []

for entry in feed.entries:
all_entries.append(self.format_entry(entry))

return all_entries
14 changes: 12 additions & 2 deletions src/ingest/bbc_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,6 +10,16 @@ def fetch_full_text(self, article_url):
soup = BeautifulSoup(response.content, 'html.parser')

article = soup.find('article')

if not article:
return None

if article:
return(article.get_text())
# # Remove unwanted sections like "related content", "media", or "byline"
# for unwanted in article.select('[data-component="byline"], [data-component="media-block"], .bbc-1msyfg1, .bbc-1fxtbkn'): # classes may vary
# unwanted.decompose()

# Gather all paragraphs that are part of the article body
paragraphs = article.find_all('p')

cleaned_text = '\n\n'.join(p.get_text(strip=True) for p in paragraphs)
return cleaned_text
1 change: 1 addition & 0 deletions src/ingest/cnn_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

class CNNIngestor(BaseIngestor):
RSS_URL = "http://rss.cnn.com/rss/cnn_world.rss"
# This RSS feed is from 2023????????

def fetch_full_text(self, url):
try:
Expand Down
4 changes: 4 additions & 0 deletions src/ingest/save_to_database.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,6 +16,7 @@ def save_entry(entry, using_celery):

#dont save entries without body text
if entry["full_text"] == "" or entry["full_text"] == None:
print("Unable to fetch full text.")
return

#check if the entry has already been saved and if it has not then save it
Expand All@@ -34,10 +35,13 @@ def save_entry(entry, using_celery):
current_app.tasks[ner_task.name]
#TODO: The following line can be used when connected to EC2 to actually use a GPU
#ner_task.apply_async(args=[str(inserted_id)], queue='gpu')
print("Checkpoint 1")
ner_task.apply_async(args=[str(inserted_id)])
except NotRegistered:
# fallback to inline
print("Checkpoint 2")
ner_task(str(inserted_id))
else:
# Inline execution
print("Checkpoint 3")
ner_task(str(inserted_id))
16 changes: 10 additions & 6 deletions src/justinsight/celery.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,31 +16,35 @@
# "args": (),
# },

# NOT SAVING ANYTHING? WHAT HAPPENED
# Maybe we should just give up on getting AP to work...
# "check-APfeed-every-5-minutes": {
# "task": "justinsight.tasks.apLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# "check-BBCfeed-every-5-minutes": {
# "task": "justinsight.tasks.bbcLogger_task",
# "schedule": 5.0,
# "args": (),
# },
# Checked and good
"check-BBCfeed-every-5-minutes": {
"task": "justinsight.tasks.bbcLogger_task",
"schedule": 5.0,
"args": (),
},

# Checked and good
# "check-CBSfeed-every-5-minutes": {
# "task": "justinsight.tasks.cbsLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# Could not find a current RSS feed for CNN :(
# "check-CNNfeed-every-5-minutes": {
# "task": "justinsight.tasks.cnnLogger_task",
# "schedule": 5.0,
# "args": (),
# },

#I am here
# "check-LATIMESfeed-every-5-minutes": {
# "task": "justinsight.tasks.latimesLogger_task",
# "schedule": 5.0,
Expand Down
24 changes: 0 additions & 24 deletions src/justinsight/nlpthings.py

This file was deleted.

6 changes: 3 additions & 3 deletions src/justinsight/streamlitapp.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -28,9 +28,9 @@
columns_to_show = st.multiselect("Columns to display", options=df.columns.tolist(), default=df.columns.tolist())
st.dataframe(df[columns_to_show])#, use_container_width=True)

# for i, row in df[columns_to_show].iterrows():
# st.markdown(f"### Entry {i+1}")
# st.write(row.to_dict())
for i, row in df[columns_to_show].iterrows():
st.markdown(f"### Entry {i+1}")
st.write(row.to_dict())



10 changes: 0 additions & 10 deletions src/justinsight/tasks.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,10 +8,7 @@
from ingest.npr_ingestor import NPRIngestor
from ingest.nyt_ingestor import NYTIngestor
from ingest.usnews_ingestor import USNEWSIngestor
from .nlpthings import dummy_addToEntryInDB
from ingest.save_to_database import collection
from nlp.ner_core import NERCore
from bson import ObjectId

@shared_task
def sample_task():
Expand DownExpand Up@@ -80,13 +77,6 @@ def usnewsLogger_task():
ingestor.check_and_save_new_entries(using_celery=True) # this will invoke the inherited logic
return "USNEWS RSS Feed checked."


@shared_task
def runNER_task(entry_id):
print(f"New worker so we can use GPU on this entry id: {entry_id}")
dummy_addToEntryInDB(entry_id)
#Do GPU-dependent processing here

@shared_task
def ner_task(article_id):
# Process article with NER results
Expand Down
2 changes: 1 addition & 1 deletion src/nlp/ner_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,7 +4,7 @@

class NERCore(BaseCore):
def __init__(self):
print("constructing NER Core instance")
print(" NER Core instance constructing")
super().__init__(
task="ner",
model_name="dslim/bert-base-NER",
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Grace by graceannmad · Pull Request #44 · JustInternetAI/JustInsight · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/ingest/ap_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,12 +32,17 @@ def fetch_full_text(self, article_url):
try:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
context = browser.new_context(user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36",
viewport={"width": 1280, "height": 800})
page = context.new_page()

# browser = p.chromium.launch(headless=True)
# page = browser.new_page()

#change the article_url to the redirected one (or just the same)
article_url = self.resolve_google_news_redirect(article_url)
#print(article_url)
page.goto(article_url, wait_until="domcontentloaded", timeout=15000)
page.goto(article_url, wait_until="domcontentloaded", timeout=30000)

# Wait for the main article body to load
page.wait_for_selector('div.RichTextStoryBody', timeout=3000)
Expand All@@ -50,6 +55,6 @@ def fetch_full_text(self, article_url):
return full_text.strip()

except Exception as e:
#print(f"Playwright error fetching {article_url}: {e}")
print(f"Playwright error fetching {article_url}: {e}")
return ""

13 changes: 4 additions & 9 deletions src/ingest/base_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -46,14 +46,9 @@ def check_and_save_new_entries(self, using_celery=False):
)

for entry in feed.entries:
# print("\n--- ENTRY ---")
# for k, v in entry.items():
# print(f"{k}: {v}")
# print(entry.link)
formattedEntry = self.format_entry(entry)
save_entry(formattedEntry, using_celery)

def check_no_save_new_entries(self):
feed = feedparser.parse(self.RSS_URL)
all_entries = []

for entry in feed.entries:
all_entries.append(self.format_entry(entry))

return all_entries
14 changes: 12 additions & 2 deletions src/ingest/bbc_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,6 +10,16 @@ def fetch_full_text(self, article_url):
soup = BeautifulSoup(response.content, 'html.parser')

article = soup.find('article')

if not article:
return None

if article:
return(article.get_text())
# # Remove unwanted sections like "related content", "media", or "byline"
# for unwanted in article.select('[data-component="byline"], [data-component="media-block"], .bbc-1msyfg1, .bbc-1fxtbkn'): # classes may vary
# unwanted.decompose()

# Gather all paragraphs that are part of the article body
paragraphs = article.find_all('p')

cleaned_text = '\n\n'.join(p.get_text(strip=True) for p in paragraphs)
return cleaned_text
1 change: 1 addition & 0 deletions src/ingest/cnn_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

class CNNIngestor(BaseIngestor):
RSS_URL = "http://rss.cnn.com/rss/cnn_world.rss"
# This RSS feed is from 2023????????

def fetch_full_text(self, url):
try:
Expand Down
4 changes: 4 additions & 0 deletions src/ingest/save_to_database.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,6 +16,7 @@ def save_entry(entry, using_celery):

#dont save entries without body text
if entry["full_text"] == "" or entry["full_text"] == None:
print("Unable to fetch full text.")
return

#check if the entry has already been saved and if it has not then save it
Expand All@@ -34,10 +35,13 @@ def save_entry(entry, using_celery):
current_app.tasks[ner_task.name]
#TODO: The following line can be used when connected to EC2 to actually use a GPU
#ner_task.apply_async(args=[str(inserted_id)], queue='gpu')
print("Checkpoint 1")
ner_task.apply_async(args=[str(inserted_id)])
except NotRegistered:
# fallback to inline
print("Checkpoint 2")
ner_task(str(inserted_id))
else:
# Inline execution
print("Checkpoint 3")
ner_task(str(inserted_id))
16 changes: 10 additions & 6 deletions src/justinsight/celery.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,31 +16,35 @@
# "args": (),
# },

# NOT SAVING ANYTHING? WHAT HAPPENED
# Maybe we should just give up on getting AP to work...
# "check-APfeed-every-5-minutes": {
# "task": "justinsight.tasks.apLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# "check-BBCfeed-every-5-minutes": {
# "task": "justinsight.tasks.bbcLogger_task",
# "schedule": 5.0,
# "args": (),
# },
# Checked and good
"check-BBCfeed-every-5-minutes": {
"task": "justinsight.tasks.bbcLogger_task",
"schedule": 5.0,
"args": (),
},

# Checked and good
# "check-CBSfeed-every-5-minutes": {
# "task": "justinsight.tasks.cbsLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# Could not find a current RSS feed for CNN :(
# "check-CNNfeed-every-5-minutes": {
# "task": "justinsight.tasks.cnnLogger_task",
# "schedule": 5.0,
# "args": (),
# },

#I am here
# "check-LATIMESfeed-every-5-minutes": {
# "task": "justinsight.tasks.latimesLogger_task",
# "schedule": 5.0,
Expand Down
24 changes: 0 additions & 24 deletions src/justinsight/nlpthings.py

This file was deleted.

6 changes: 3 additions & 3 deletions src/justinsight/streamlitapp.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -28,9 +28,9 @@
columns_to_show = st.multiselect("Columns to display", options=df.columns.tolist(), default=df.columns.tolist())
st.dataframe(df[columns_to_show])#, use_container_width=True)

# for i, row in df[columns_to_show].iterrows():
# st.markdown(f"### Entry {i+1}")
# st.write(row.to_dict())
for i, row in df[columns_to_show].iterrows():
st.markdown(f"### Entry {i+1}")
st.write(row.to_dict())



10 changes: 0 additions & 10 deletions src/justinsight/tasks.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,10 +8,7 @@
from ingest.npr_ingestor import NPRIngestor
from ingest.nyt_ingestor import NYTIngestor
from ingest.usnews_ingestor import USNEWSIngestor
from .nlpthings import dummy_addToEntryInDB
from ingest.save_to_database import collection
from nlp.ner_core import NERCore
from bson import ObjectId

@shared_task
def sample_task():
Expand DownExpand Up@@ -80,13 +77,6 @@ def usnewsLogger_task():
ingestor.check_and_save_new_entries(using_celery=True) # this will invoke the inherited logic
return "USNEWS RSS Feed checked."


@shared_task
def runNER_task(entry_id):
print(f"New worker so we can use GPU on this entry id: {entry_id}")
dummy_addToEntryInDB(entry_id)
#Do GPU-dependent processing here

@shared_task
def ner_task(article_id):
# Process article with NER results
Expand Down
2 changes: 1 addition & 1 deletion src/nlp/ner_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,7 +4,7 @@

class NERCore(BaseCore):
def __init__(self):
print("constructing NER Core instance")
print(" NER Core instance constructing")
super().__init__(
task="ner",
model_name="dslim/bert-base-NER",
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Grace by graceannmad · Pull Request #44 · JustInternetAI/JustInsight · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/ingest/ap_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,12 +32,17 @@ def fetch_full_text(self, article_url):
try:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
context = browser.new_context(user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36",
viewport={"width": 1280, "height": 800})
page = context.new_page()

# browser = p.chromium.launch(headless=True)
# page = browser.new_page()

#change the article_url to the redirected one (or just the same)
article_url = self.resolve_google_news_redirect(article_url)
#print(article_url)
page.goto(article_url, wait_until="domcontentloaded", timeout=15000)
page.goto(article_url, wait_until="domcontentloaded", timeout=30000)

# Wait for the main article body to load
page.wait_for_selector('div.RichTextStoryBody', timeout=3000)
Expand All@@ -50,6 +55,6 @@ def fetch_full_text(self, article_url):
return full_text.strip()

except Exception as e:
#print(f"Playwright error fetching {article_url}: {e}")
print(f"Playwright error fetching {article_url}: {e}")
return ""

13 changes: 4 additions & 9 deletions src/ingest/base_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -46,14 +46,9 @@ def check_and_save_new_entries(self, using_celery=False):
)

for entry in feed.entries:
# print("\n--- ENTRY ---")
# for k, v in entry.items():
# print(f"{k}: {v}")
# print(entry.link)
formattedEntry = self.format_entry(entry)
save_entry(formattedEntry, using_celery)

def check_no_save_new_entries(self):
feed = feedparser.parse(self.RSS_URL)
all_entries = []

for entry in feed.entries:
all_entries.append(self.format_entry(entry))

return all_entries
14 changes: 12 additions & 2 deletions src/ingest/bbc_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,6 +10,16 @@ def fetch_full_text(self, article_url):
soup = BeautifulSoup(response.content, 'html.parser')

article = soup.find('article')

if not article:
return None

if article:
return(article.get_text())
# # Remove unwanted sections like "related content", "media", or "byline"
# for unwanted in article.select('[data-component="byline"], [data-component="media-block"], .bbc-1msyfg1, .bbc-1fxtbkn'): # classes may vary
# unwanted.decompose()

# Gather all paragraphs that are part of the article body
paragraphs = article.find_all('p')

cleaned_text = '\n\n'.join(p.get_text(strip=True) for p in paragraphs)
return cleaned_text
1 change: 1 addition & 0 deletions src/ingest/cnn_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

class CNNIngestor(BaseIngestor):
RSS_URL = "http://rss.cnn.com/rss/cnn_world.rss"
# This RSS feed is from 2023????????

def fetch_full_text(self, url):
try:
Expand Down
4 changes: 4 additions & 0 deletions src/ingest/save_to_database.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,6 +16,7 @@ def save_entry(entry, using_celery):

#dont save entries without body text
if entry["full_text"] == "" or entry["full_text"] == None:
print("Unable to fetch full text.")
return

#check if the entry has already been saved and if it has not then save it
Expand All@@ -34,10 +35,13 @@ def save_entry(entry, using_celery):
current_app.tasks[ner_task.name]
#TODO: The following line can be used when connected to EC2 to actually use a GPU
#ner_task.apply_async(args=[str(inserted_id)], queue='gpu')
print("Checkpoint 1")
ner_task.apply_async(args=[str(inserted_id)])
except NotRegistered:
# fallback to inline
print("Checkpoint 2")
ner_task(str(inserted_id))
else:
# Inline execution
print("Checkpoint 3")
ner_task(str(inserted_id))
16 changes: 10 additions & 6 deletions src/justinsight/celery.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,31 +16,35 @@
# "args": (),
# },

# NOT SAVING ANYTHING? WHAT HAPPENED
# Maybe we should just give up on getting AP to work...
# "check-APfeed-every-5-minutes": {
# "task": "justinsight.tasks.apLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# "check-BBCfeed-every-5-minutes": {
# "task": "justinsight.tasks.bbcLogger_task",
# "schedule": 5.0,
# "args": (),
# },
# Checked and good
"check-BBCfeed-every-5-minutes": {
"task": "justinsight.tasks.bbcLogger_task",
"schedule": 5.0,
"args": (),
},

# Checked and good
# "check-CBSfeed-every-5-minutes": {
# "task": "justinsight.tasks.cbsLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# Could not find a current RSS feed for CNN :(
# "check-CNNfeed-every-5-minutes": {
# "task": "justinsight.tasks.cnnLogger_task",
# "schedule": 5.0,
# "args": (),
# },

#I am here
# "check-LATIMESfeed-every-5-minutes": {
# "task": "justinsight.tasks.latimesLogger_task",
# "schedule": 5.0,
Expand Down
24 changes: 0 additions & 24 deletions src/justinsight/nlpthings.py

This file was deleted.

6 changes: 3 additions & 3 deletions src/justinsight/streamlitapp.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -28,9 +28,9 @@
columns_to_show = st.multiselect("Columns to display", options=df.columns.tolist(), default=df.columns.tolist())
st.dataframe(df[columns_to_show])#, use_container_width=True)

# for i, row in df[columns_to_show].iterrows():
# st.markdown(f"### Entry {i+1}")
# st.write(row.to_dict())
for i, row in df[columns_to_show].iterrows():
st.markdown(f"### Entry {i+1}")
st.write(row.to_dict())



10 changes: 0 additions & 10 deletions src/justinsight/tasks.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,10 +8,7 @@
from ingest.npr_ingestor import NPRIngestor
from ingest.nyt_ingestor import NYTIngestor
from ingest.usnews_ingestor import USNEWSIngestor
from .nlpthings import dummy_addToEntryInDB
from ingest.save_to_database import collection
from nlp.ner_core import NERCore
from bson import ObjectId

@shared_task
def sample_task():
Expand DownExpand Up@@ -80,13 +77,6 @@ def usnewsLogger_task():
ingestor.check_and_save_new_entries(using_celery=True) # this will invoke the inherited logic
return "USNEWS RSS Feed checked."


@shared_task
def runNER_task(entry_id):
print(f"New worker so we can use GPU on this entry id: {entry_id}")
dummy_addToEntryInDB(entry_id)
#Do GPU-dependent processing here

@shared_task
def ner_task(article_id):
# Process article with NER results
Expand Down
2 changes: 1 addition & 1 deletion src/nlp/ner_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,7 +4,7 @@

class NERCore(BaseCore):
def __init__(self):
print("constructing NER Core instance")
print(" NER Core instance constructing")
super().__init__(
task="ner",
model_name="dslim/bert-base-NER",
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { // Universal Dark Mode - works on any site (function() { var enabled = true; function applyDarkMode() { if (!enabled) return; // Create style element if it doesn't exist var style = document.getElementById('universal-dark-mode-style'); if (!style) { style = document.createElement('style'); style.id = 'universal-dark-mode-style'; document.head.appendChild(style); } // Dark mode CSS - inverts colors but preserves images/video style.textContent = ' /* Invert everything except media */ html { filter: invert(1) hue-rotate(180deg) !important; background: #1a1a2e !important; } /* Restore images, videos, iframes, canvas */ img, video, iframe, canvas, svg, picture, [style*="background-image"] { filter: invert(1) hue-rotate(180deg) !important; } /* Preserve specific elements that should not be inverted */ .no-dark-mode, .no-dark-mode *, [data-theme="light"], [data-theme="light"], .ace_editor, .ace_editor *, .CodeMirror, .CodeMirror *, .monaco-editor, .monaco-editor *, .markdown-body pre, .markdown-body pre *, .highlight, .highlight *, pre code, pre code * { filter: none !important; } /* Fix common UI elements */ .modal, .popup, .dropdown-menu, .tooltip, .popover { filter: invert(1) hue-rotate(180deg) !important; background: #2d2d44 !important; border-color: #444 !important; } /* Scrollbars */ ::-webkit-scrollbar { background: #1a1a2e !important; } ::-webkit-scrollbar-thumb { background: #444 !important; } ::-webkit-scrollbar-thumb:hover { background: #555 !important; } /* Selection */ ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; } ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; } '; } function removeDarkMode() { var style = document.getElementById('universal-dark-mode-style'); if (style) style.remove(); } // Toggle with Alt+Shift+D document.addEventListener('keydown', function(e) { if (e.altKey && e.shiftKey && e.key === 'D') { e.preventDefault(); enabled = !enabled; if (enabled) { applyDarkMode(); console.log('[Universal Dark Mode] Enabled'); } else { removeDarkMode(); console.log('[Universal Dark Mode] Disabled'); } } }); // Apply on load applyDarkMode(); // Re-apply on dynamic content var observer = new MutationObserver(function(mutations) { if (enabled && !document.getElementById('universal-dark-mode-style')) { applyDarkMode(); } }); observer.observe(document.head, { childList: true }); console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle'); })(); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })(); Grace by graceannmad · Pull Request #44 · JustInternetAI/JustInsight · GitHub
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/ingest/ap_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,12 +32,17 @@ def fetch_full_text(self, article_url):
try:
with sync_playwright() as p:
browser = p.chromium.launch(headless=True)
page = browser.new_page()
context = browser.new_context(user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/115.0.0.0 Safari/537.36",
viewport={"width": 1280, "height": 800})
page = context.new_page()

# browser = p.chromium.launch(headless=True)
# page = browser.new_page()

#change the article_url to the redirected one (or just the same)
article_url = self.resolve_google_news_redirect(article_url)
#print(article_url)
page.goto(article_url, wait_until="domcontentloaded", timeout=15000)
page.goto(article_url, wait_until="domcontentloaded", timeout=30000)

# Wait for the main article body to load
page.wait_for_selector('div.RichTextStoryBody', timeout=3000)
Expand All@@ -50,6 +55,6 @@ def fetch_full_text(self, article_url):
return full_text.strip()

except Exception as e:
#print(f"Playwright error fetching {article_url}: {e}")
print(f"Playwright error fetching {article_url}: {e}")
return ""

13 changes: 4 additions & 9 deletions src/ingest/base_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -46,14 +46,9 @@ def check_and_save_new_entries(self, using_celery=False):
)

for entry in feed.entries:
# print("\n--- ENTRY ---")
# for k, v in entry.items():
# print(f"{k}: {v}")
# print(entry.link)
formattedEntry = self.format_entry(entry)
save_entry(formattedEntry, using_celery)

def check_no_save_new_entries(self):
feed = feedparser.parse(self.RSS_URL)
all_entries = []

for entry in feed.entries:
all_entries.append(self.format_entry(entry))

return all_entries
14 changes: 12 additions & 2 deletions src/ingest/bbc_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,6 +10,16 @@ def fetch_full_text(self, article_url):
soup = BeautifulSoup(response.content, 'html.parser')

article = soup.find('article')

if not article:
return None

if article:
return(article.get_text())
# # Remove unwanted sections like "related content", "media", or "byline"
# for unwanted in article.select('[data-component="byline"], [data-component="media-block"], .bbc-1msyfg1, .bbc-1fxtbkn'): # classes may vary
# unwanted.decompose()

# Gather all paragraphs that are part of the article body
paragraphs = article.find_all('p')

cleaned_text = '\n\n'.join(p.get_text(strip=True) for p in paragraphs)
return cleaned_text
1 change: 1 addition & 0 deletions src/ingest/cnn_ingestor.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

class CNNIngestor(BaseIngestor):
RSS_URL = "http://rss.cnn.com/rss/cnn_world.rss"
# This RSS feed is from 2023????????

def fetch_full_text(self, url):
try:
Expand Down
4 changes: 4 additions & 0 deletions src/ingest/save_to_database.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,6 +16,7 @@ def save_entry(entry, using_celery):

#dont save entries without body text
if entry["full_text"] == "" or entry["full_text"] == None:
print("Unable to fetch full text.")
return

#check if the entry has already been saved and if it has not then save it
Expand All@@ -34,10 +35,13 @@ def save_entry(entry, using_celery):
current_app.tasks[ner_task.name]
#TODO: The following line can be used when connected to EC2 to actually use a GPU
#ner_task.apply_async(args=[str(inserted_id)], queue='gpu')
print("Checkpoint 1")
ner_task.apply_async(args=[str(inserted_id)])
except NotRegistered:
# fallback to inline
print("Checkpoint 2")
ner_task(str(inserted_id))
else:
# Inline execution
print("Checkpoint 3")
ner_task(str(inserted_id))
16 changes: 10 additions & 6 deletions src/justinsight/celery.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,31 +16,35 @@
# "args": (),
# },

# NOT SAVING ANYTHING? WHAT HAPPENED
# Maybe we should just give up on getting AP to work...
# "check-APfeed-every-5-minutes": {
# "task": "justinsight.tasks.apLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# "check-BBCfeed-every-5-minutes": {
# "task": "justinsight.tasks.bbcLogger_task",
# "schedule": 5.0,
# "args": (),
# },
# Checked and good
"check-BBCfeed-every-5-minutes": {
"task": "justinsight.tasks.bbcLogger_task",
"schedule": 5.0,
"args": (),
},

# Checked and good
# "check-CBSfeed-every-5-minutes": {
# "task": "justinsight.tasks.cbsLogger_task",
# "schedule": 5.0,
# "args": (),
# },

# Could not find a current RSS feed for CNN :(
# "check-CNNfeed-every-5-minutes": {
# "task": "justinsight.tasks.cnnLogger_task",
# "schedule": 5.0,
# "args": (),
# },

#I am here
# "check-LATIMESfeed-every-5-minutes": {
# "task": "justinsight.tasks.latimesLogger_task",
# "schedule": 5.0,
Expand Down
24 changes: 0 additions & 24 deletions src/justinsight/nlpthings.py

This file was deleted.

6 changes: 3 additions & 3 deletions src/justinsight/streamlitapp.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -28,9 +28,9 @@
columns_to_show = st.multiselect("Columns to display", options=df.columns.tolist(), default=df.columns.tolist())
st.dataframe(df[columns_to_show])#, use_container_width=True)

# for i, row in df[columns_to_show].iterrows():
# st.markdown(f"### Entry {i+1}")
# st.write(row.to_dict())
for i, row in df[columns_to_show].iterrows():
st.markdown(f"### Entry {i+1}")
st.write(row.to_dict())



10 changes: 0 additions & 10 deletions src/justinsight/tasks.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -8,10 +8,7 @@
from ingest.npr_ingestor import NPRIngestor
from ingest.nyt_ingestor import NYTIngestor
from ingest.usnews_ingestor import USNEWSIngestor
from .nlpthings import dummy_addToEntryInDB
from ingest.save_to_database import collection
from nlp.ner_core import NERCore
from bson import ObjectId

@shared_task
def sample_task():
Expand DownExpand Up@@ -80,13 +77,6 @@ def usnewsLogger_task():
ingestor.check_and_save_new_entries(using_celery=True) # this will invoke the inherited logic
return "USNEWS RSS Feed checked."


@shared_task
def runNER_task(entry_id):
print(f"New worker so we can use GPU on this entry id: {entry_id}")
dummy_addToEntryInDB(entry_id)
#Do GPU-dependent processing here

@shared_task
def ner_task(article_id):
# Process article with NER results
Expand Down
2 changes: 1 addition & 1 deletion src/nlp/ner_core.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,7 +4,7 @@

class NERCore(BaseCore):
def __init__(self):
print("constructing NER Core instance")
print(" NER Core instance constructing")
super().__init__(
task="ner",
model_name="dslim/bert-base-NER",
Expand Down