mirror of https://gitlab.com/bashrc2/epicyon
Indexing of content labels
parent
72fe30a5e5
commit
c58b7b16f4
|
|
@ -7,9 +7,32 @@ __email__ = "bob@libreserver.org"
|
||||||
__status__ = "Production"
|
__status__ = "Production"
|
||||||
__module_group__ = "Daemon POST"
|
__module_group__ = "Daemon POST"
|
||||||
|
|
||||||
|
import os
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from src.utils import string_contains
|
||||||
|
from src.utils import data_dir
|
||||||
|
from src.utils import local_actor_url
|
||||||
|
from src.utils import file_last_modified
|
||||||
|
from src.utils import acct_dir
|
||||||
|
from src.utils import valid_content_label
|
||||||
|
from src.utils import remove_id_ending
|
||||||
from src.utils import remove_html
|
from src.utils import remove_html
|
||||||
from src.utils import resembles_url
|
from src.utils import resembles_url
|
||||||
from src.utils import has_object_dict
|
from src.utils import has_object_dict
|
||||||
|
from src.data import load_line
|
||||||
|
from src.data import load_string
|
||||||
|
from src.data import is_a_dir
|
||||||
|
from src.data import makedir
|
||||||
|
from src.data import is_a_file
|
||||||
|
from src.data import save_string
|
||||||
|
from src.maps import get_map_links_from_post_content
|
||||||
|
from src.maps import get_location_from_post
|
||||||
|
from src.maps import geocoords_from_map_link
|
||||||
|
from src.maps import add_label_map_links
|
||||||
|
from src.timeFunctions import date_utcnow
|
||||||
|
from src.timeFunctions import date_epoch
|
||||||
|
from src.timeFunctions import date_from_string_format
|
||||||
|
# from src.delete import remove_old_labels
|
||||||
|
|
||||||
MAX_CONTENT_LABELS = 10
|
MAX_CONTENT_LABELS = 10
|
||||||
|
|
||||||
|
|
@ -51,6 +74,8 @@ def get_labels_from_json(json_object: {}) -> []:
|
||||||
else:
|
else:
|
||||||
labels_list = tag_dict['value'].split('/')
|
labels_list = tag_dict['value'].split('/')
|
||||||
for label_str in labels_list:
|
for label_str in labels_list:
|
||||||
|
if not valid_content_label(label_str):
|
||||||
|
break
|
||||||
label_str = label_str.strip()
|
label_str = label_str.strip()
|
||||||
label_str = remove_html(label_str)
|
label_str = remove_html(label_str)
|
||||||
if label_str:
|
if label_str:
|
||||||
|
|
@ -65,6 +90,8 @@ def get_labels_from_json(json_object: {}) -> []:
|
||||||
label_str = tag_dict['value'].strip()
|
label_str = tag_dict['value'].strip()
|
||||||
label_str = remove_html(label_str)
|
label_str = remove_html(label_str)
|
||||||
if label_str:
|
if label_str:
|
||||||
|
if not valid_content_label(label_str):
|
||||||
|
break
|
||||||
if 'href' in tag_dict:
|
if 'href' in tag_dict:
|
||||||
if isinstance(tag_dict['href'], str):
|
if isinstance(tag_dict['href'], str):
|
||||||
if resembles_url(tag_dict['href']):
|
if resembles_url(tag_dict['href']):
|
||||||
|
|
@ -77,6 +104,8 @@ def get_labels_from_json(json_object: {}) -> []:
|
||||||
elif tag_dict['type'] == 'Label':
|
elif tag_dict['type'] == 'Label':
|
||||||
label_str = remove_html(tag_dict['name'])
|
label_str = remove_html(tag_dict['name'])
|
||||||
if label_str:
|
if label_str:
|
||||||
|
if not valid_content_label(label_str):
|
||||||
|
break
|
||||||
if 'href' in tag_dict:
|
if 'href' in tag_dict:
|
||||||
if isinstance(tag_dict['href'], str):
|
if isinstance(tag_dict['href'], str):
|
||||||
if resembles_url(tag_dict['href']):
|
if resembles_url(tag_dict['href']):
|
||||||
|
|
@ -94,6 +123,8 @@ def labels_list_html(labels_list: []) -> str:
|
||||||
"""
|
"""
|
||||||
labels_str = ''
|
labels_str = ''
|
||||||
for label in labels_list:
|
for label in labels_list:
|
||||||
|
if not valid_content_label(label):
|
||||||
|
continue
|
||||||
label_url = ''
|
label_url = ''
|
||||||
if '###' in label:
|
if '###' in label:
|
||||||
label_url = label.split('###')[1]
|
label_url = label.split('###')[1]
|
||||||
|
|
@ -101,7 +132,10 @@ def labels_list_html(labels_list: []) -> str:
|
||||||
if labels_str:
|
if labels_str:
|
||||||
labels_str += ' '
|
labels_str += ' '
|
||||||
if not label_url:
|
if not label_url:
|
||||||
labels_str += '<mark>' + label + '</mark>'
|
label_in_path = label.replace(' ', '_')
|
||||||
|
labels_str += '<mark><a href="/labels/' + label_in_path + \
|
||||||
|
'" target="_blank" rel="nofollow noopener noreferrer">' + \
|
||||||
|
label + '</a></mark>'
|
||||||
else:
|
else:
|
||||||
labels_str += '<mark><a href="' + label_url + \
|
labels_str += '<mark><a href="' + label_url + \
|
||||||
'" target="_blank" rel="nofollow noopener noreferrer">' + \
|
'" target="_blank" rel="nofollow noopener noreferrer">' + \
|
||||||
|
|
@ -118,6 +152,8 @@ def get_actor_content_labels(actor_json: {}) -> str:
|
||||||
labels_list: list[str] = get_labels_from_json(actor_json)
|
labels_list: list[str] = get_labels_from_json(actor_json)
|
||||||
labels_str = ''
|
labels_str = ''
|
||||||
for lbl in labels_list:
|
for lbl in labels_list:
|
||||||
|
if '###' in lbl:
|
||||||
|
lbl = lbl.split('###')[0]
|
||||||
if labels_str:
|
if labels_str:
|
||||||
labels_str += ', '
|
labels_str += ', '
|
||||||
labels_str += lbl
|
labels_str += lbl
|
||||||
|
|
@ -181,3 +217,268 @@ def set_post_content_labels(post_json_object: {}, labels: str) -> None:
|
||||||
'value': labels
|
'value': labels
|
||||||
}
|
}
|
||||||
obj['tag'].append(labels_dict)
|
obj['tag'].append(labels_dict)
|
||||||
|
|
||||||
|
|
||||||
|
def _store_content_label(nickname: str,
|
||||||
|
label: str, labels_dir: str, post_url: str,
|
||||||
|
map_links: [], published: str,
|
||||||
|
labels_maps_dir: str) -> bool:
|
||||||
|
"""stores an individual content_label
|
||||||
|
"""
|
||||||
|
if not valid_content_label(label):
|
||||||
|
return False
|
||||||
|
labels_filename = labels_dir + '/' + label + '.txt'
|
||||||
|
days_diff = date_utcnow() - date_epoch()
|
||||||
|
days_since_epoch = days_diff.days
|
||||||
|
label_line = \
|
||||||
|
str(days_since_epoch) + ' ' + nickname + ' ' + post_url + '\n'
|
||||||
|
if map_links and published:
|
||||||
|
add_label_map_links(labels_maps_dir, label, map_links,
|
||||||
|
published, post_url)
|
||||||
|
label_added: bool = False
|
||||||
|
if not is_a_file(labels_filename):
|
||||||
|
if save_string(label_line, labels_filename,
|
||||||
|
'EX: _store_content_label unable to write ' +
|
||||||
|
labels_filename):
|
||||||
|
label_added = True
|
||||||
|
else:
|
||||||
|
content = load_string(labels_filename,
|
||||||
|
'EX: _store_content_label failed to read ' +
|
||||||
|
labels_filename)
|
||||||
|
if content is None:
|
||||||
|
content: str = ''
|
||||||
|
if post_url not in content:
|
||||||
|
content = label_line + content
|
||||||
|
if save_string(content, labels_filename,
|
||||||
|
'EX: Failed to write entry to labels file ' +
|
||||||
|
labels_filename + ' [ex]'):
|
||||||
|
label_added = True
|
||||||
|
|
||||||
|
if not label_added:
|
||||||
|
return False
|
||||||
|
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def _html_labels_swarm(base_dir: str, actor: str) -> str:
|
||||||
|
"""Returns a labels swarm of today's labels
|
||||||
|
"""
|
||||||
|
max_label_length = 42
|
||||||
|
curr_time = date_utcnow()
|
||||||
|
prev_time_epoch = date_epoch()
|
||||||
|
days_since_epoch = (curr_time - prev_time_epoch).days
|
||||||
|
days_since_epoch_str = str(days_since_epoch) + ' '
|
||||||
|
days_since_epoch_str2 = str(days_since_epoch - 1) + ' '
|
||||||
|
recently = days_since_epoch - 1
|
||||||
|
labels_swarm: list[str] = []
|
||||||
|
domain_histogram = {}
|
||||||
|
|
||||||
|
# Load the blocked labels into memory.
|
||||||
|
# This avoids needing to repeatedly load the blocked file for each label
|
||||||
|
blocked_str: str = ''
|
||||||
|
global_blocking_filename = data_dir(base_dir) + '/blocking.txt'
|
||||||
|
if is_a_file(global_blocking_filename):
|
||||||
|
blocked_str = \
|
||||||
|
load_string(global_blocking_filename,
|
||||||
|
'EX: _html_labels_swarm unable to read ' +
|
||||||
|
global_blocking_filename)
|
||||||
|
if blocked_str is None:
|
||||||
|
blocked_str: str = ''
|
||||||
|
|
||||||
|
for _, _, files in os.walk(base_dir + '/labels'):
|
||||||
|
for fname in files:
|
||||||
|
if not fname.endswith('.txt'):
|
||||||
|
continue
|
||||||
|
labels_filename = os.path.join(base_dir + '/labels', fname)
|
||||||
|
if not is_a_file(labels_filename):
|
||||||
|
continue
|
||||||
|
|
||||||
|
# get last modified datetime
|
||||||
|
mod_time_since_epoc = os.path.getmtime(labels_filename)
|
||||||
|
last_modified_date = \
|
||||||
|
datetime.fromtimestamp(mod_time_since_epoc,
|
||||||
|
timezone.utc)
|
||||||
|
file_days_since_epoch = \
|
||||||
|
(last_modified_date - prev_time_epoch).days
|
||||||
|
|
||||||
|
# check if the file was last modified within the previous
|
||||||
|
# two days
|
||||||
|
if file_days_since_epoch < recently:
|
||||||
|
continue
|
||||||
|
|
||||||
|
label = fname.replace('.txt', '')
|
||||||
|
if len(label) > max_label_length:
|
||||||
|
continue
|
||||||
|
if string_contains(label, ['#', '&', '"', "'"]):
|
||||||
|
continue
|
||||||
|
if label + '\n' in blocked_str:
|
||||||
|
continue
|
||||||
|
|
||||||
|
last_label = \
|
||||||
|
load_line(labels_filename,
|
||||||
|
'EX: _html_labels_swarm unable to read 2 ' +
|
||||||
|
labels_filename)
|
||||||
|
if last_label is None:
|
||||||
|
continue
|
||||||
|
if not last_label.startswith(days_since_epoch_str):
|
||||||
|
if not last_label.startswith(days_since_epoch_str2):
|
||||||
|
continue
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(labels_filename, 'r', encoding='utf-8') as fp_labels:
|
||||||
|
while True:
|
||||||
|
line = fp_labels.readline()
|
||||||
|
if not line:
|
||||||
|
break
|
||||||
|
if ' ' not in line:
|
||||||
|
break
|
||||||
|
sections: list[str] = line.split(' ')
|
||||||
|
if len(sections) != 3:
|
||||||
|
break
|
||||||
|
post_days_since_epoch_str = sections[0]
|
||||||
|
if not post_days_since_epoch_str.isdigit():
|
||||||
|
break
|
||||||
|
post_days_since_epoch = int(post_days_since_epoch_str)
|
||||||
|
if post_days_since_epoch < recently:
|
||||||
|
break
|
||||||
|
post_url = sections[2]
|
||||||
|
if '##' not in post_url:
|
||||||
|
break
|
||||||
|
post_domain = post_url.split('##')[1]
|
||||||
|
if '#' in post_domain:
|
||||||
|
post_domain = post_domain.split('#')[0]
|
||||||
|
|
||||||
|
if domain_histogram.get(post_domain):
|
||||||
|
domain_histogram[post_domain] = \
|
||||||
|
domain_histogram[post_domain] + 1
|
||||||
|
else:
|
||||||
|
domain_histogram[post_domain] = 1
|
||||||
|
labels_swarm.append(label)
|
||||||
|
break
|
||||||
|
except OSError as exc:
|
||||||
|
print('EX: _html_labels_swarm unable to read ' +
|
||||||
|
labels_filename + ' ' + str(exc))
|
||||||
|
break
|
||||||
|
|
||||||
|
if not labels_swarm:
|
||||||
|
return ''
|
||||||
|
labels_swarm.sort()
|
||||||
|
|
||||||
|
# swarm of labels
|
||||||
|
labels_swarm_str: str = ''
|
||||||
|
for label in labels_swarm:
|
||||||
|
label_display_name = label
|
||||||
|
labels_map_filename = \
|
||||||
|
os.path.join(base_dir + '/labelsmaps', label + '.txt')
|
||||||
|
if is_a_file(labels_map_filename):
|
||||||
|
label_display_name = '📌' + label
|
||||||
|
label_in_path = label.replace(' ', '_')
|
||||||
|
labels_swarm_str += \
|
||||||
|
'<a href="' + actor + '/labels/' + label_in_path + \
|
||||||
|
'" class="hashtagswarm">' + label_display_name + '</a>\n'
|
||||||
|
|
||||||
|
labels_swarm_html = labels_swarm_str.strip() + '\n'
|
||||||
|
return labels_swarm_html
|
||||||
|
|
||||||
|
|
||||||
|
def _update_cached_labels_swarm(base_dir: str, nickname: str, domain: str,
|
||||||
|
http_prefix: str, domain_full: str) -> bool:
|
||||||
|
"""Updates the labels swarm stored as a file
|
||||||
|
"""
|
||||||
|
cached_labels_swarm_filename = \
|
||||||
|
acct_dir(base_dir, nickname, domain) + '/.labelsSwarm'
|
||||||
|
save_swarm = True
|
||||||
|
if is_a_file(cached_labels_swarm_filename):
|
||||||
|
last_modified = file_last_modified(cached_labels_swarm_filename)
|
||||||
|
modified_date = None
|
||||||
|
try:
|
||||||
|
modified_date = \
|
||||||
|
date_from_string_format(last_modified, ["%Y-%m-%dT%H:%M:%S%z"])
|
||||||
|
except BaseException:
|
||||||
|
print('EX: unable to parse last modified cache date ' +
|
||||||
|
str(last_modified))
|
||||||
|
if modified_date:
|
||||||
|
curr_date = date_utcnow()
|
||||||
|
time_diff = curr_date - modified_date
|
||||||
|
diff_mins = int(time_diff.total_seconds() / 60)
|
||||||
|
if diff_mins < 30:
|
||||||
|
# was saved recently, so don't save again
|
||||||
|
# This avoids too much disk I/O
|
||||||
|
save_swarm: bool = False
|
||||||
|
print('Not updating labels swarm')
|
||||||
|
else:
|
||||||
|
print('Updating cached labels swarm, last changed ' +
|
||||||
|
str(diff_mins) + ' minutes ago')
|
||||||
|
else:
|
||||||
|
print('WARN: no modified date for ' + str(last_modified))
|
||||||
|
if save_swarm:
|
||||||
|
actor = local_actor_url(http_prefix, nickname, domain_full)
|
||||||
|
new_swarm_str = _html_labels_swarm(base_dir, actor)
|
||||||
|
if new_swarm_str:
|
||||||
|
if save_string(new_swarm_str, cached_labels_swarm_filename,
|
||||||
|
'EX: unable to write cached labels swarm ' +
|
||||||
|
cached_labels_swarm_filename):
|
||||||
|
return True
|
||||||
|
# remove_old_labels(base_dir, 3)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def store_content_labels(base_dir: str, nickname: str, domain: str,
|
||||||
|
http_prefix: str, domain_full: str,
|
||||||
|
post_json_object: {},
|
||||||
|
session) -> None:
|
||||||
|
"""Extracts content labels from an incoming post and updates the
|
||||||
|
relevant label files.
|
||||||
|
"""
|
||||||
|
labels_list: list[str] = get_labels_from_json(post_json_object)
|
||||||
|
|
||||||
|
labels_dir = base_dir + '/labels'
|
||||||
|
|
||||||
|
# add labels directory if it doesn't exist
|
||||||
|
if not is_a_dir(labels_dir):
|
||||||
|
print('Creating content labels directory')
|
||||||
|
makedir(labels_dir)
|
||||||
|
|
||||||
|
# obtain any map links and these can be associated with labelss
|
||||||
|
# get geolocations from content
|
||||||
|
map_links: list[str] = []
|
||||||
|
published = None
|
||||||
|
if 'content' in post_json_object['object']:
|
||||||
|
published = post_json_object['object']['published']
|
||||||
|
post_content = post_json_object['object']['content']
|
||||||
|
map_links += get_map_links_from_post_content(post_content, session)
|
||||||
|
# get geolocation from labels
|
||||||
|
location_str = get_location_from_post(post_json_object)
|
||||||
|
if location_str:
|
||||||
|
# remove address if needed
|
||||||
|
if '<br><address>' in location_str:
|
||||||
|
location_str = location_str.split('<br><address>')[0].strip()
|
||||||
|
if resembles_url(location_str):
|
||||||
|
zoom, latitude, longitude = \
|
||||||
|
geocoords_from_map_link(location_str,
|
||||||
|
'openstreetmap.org', session)
|
||||||
|
if latitude and longitude and zoom and \
|
||||||
|
location_str not in map_links:
|
||||||
|
map_links.append(location_str)
|
||||||
|
labels_maps_dir = base_dir + '/labelsmaps'
|
||||||
|
if map_links:
|
||||||
|
# add labelsmaps directory if it doesn't exist
|
||||||
|
if not is_a_dir(labels_maps_dir):
|
||||||
|
print('Creating labelsmaps directory')
|
||||||
|
makedir(labels_maps_dir)
|
||||||
|
|
||||||
|
post_url = remove_id_ending(post_json_object['id'])
|
||||||
|
post_url = post_url.replace('/', '#')
|
||||||
|
labels_ctr: int = 0
|
||||||
|
for label in labels_list:
|
||||||
|
if _store_content_label(nickname,
|
||||||
|
label, labels_dir, post_url,
|
||||||
|
map_links, published,
|
||||||
|
labels_maps_dir):
|
||||||
|
labels_ctr += 1
|
||||||
|
|
||||||
|
# if some labels were found then recalculate the swarm
|
||||||
|
# ready for later display
|
||||||
|
if labels_ctr > 0:
|
||||||
|
_update_cached_labels_swarm(base_dir, nickname, domain,
|
||||||
|
http_prefix, domain_full)
|
||||||
|
|
|
||||||
|
|
@ -219,3 +219,36 @@ def remove_old_hashtags(base_dir: str, max_months: int) -> str:
|
||||||
erase_file(erase_filename,
|
erase_file(erase_filename,
|
||||||
'EX: remove_old_hashtags unable to delete ' +
|
'EX: remove_old_hashtags unable to delete ' +
|
||||||
erase_filename)
|
erase_filename)
|
||||||
|
|
||||||
|
|
||||||
|
def remove_old_labels(base_dir: str, max_months: int) -> str:
|
||||||
|
"""Remove old labels
|
||||||
|
"""
|
||||||
|
max_months: int = min(max_months, 11)
|
||||||
|
prev_date = date_from_numbers(1970, 1 + max_months, 1, 0, 0)
|
||||||
|
max_days_since_epoch: int = (date_utcnow() - prev_date).days
|
||||||
|
remove_labels: list[str] = []
|
||||||
|
|
||||||
|
for _, _, files in os.walk(base_dir + '/labels'):
|
||||||
|
for fname in files:
|
||||||
|
labels_filename: str = os.path.join(base_dir + '/labels', fname)
|
||||||
|
if not is_a_file(labels_filename):
|
||||||
|
continue
|
||||||
|
# get last modified datetime
|
||||||
|
mod_time_since_epoc = os.path.getmtime(labels_filename)
|
||||||
|
last_modified_date = \
|
||||||
|
datetime.fromtimestamp(mod_time_since_epoc,
|
||||||
|
timezone.utc)
|
||||||
|
prev_date_epoch = date_epoch()
|
||||||
|
file_days_since_epoch = \
|
||||||
|
(last_modified_date - prev_date_epoch).days
|
||||||
|
|
||||||
|
# check of the file is too old
|
||||||
|
if file_days_since_epoch < max_days_since_epoch:
|
||||||
|
remove_labels.append(labels_filename)
|
||||||
|
break
|
||||||
|
|
||||||
|
for erase_filename in remove_labels:
|
||||||
|
erase_file(erase_filename,
|
||||||
|
'EX: remove_old_labels unable to delete ' +
|
||||||
|
erase_filename)
|
||||||
|
|
|
||||||
10
src/inbox.py
10
src/inbox.py
|
|
@ -152,6 +152,7 @@ from src.data import erase_file
|
||||||
from src.data import is_a_file
|
from src.data import is_a_file
|
||||||
from src.data import is_a_dir
|
from src.data import is_a_dir
|
||||||
from src.data import makedir
|
from src.data import makedir
|
||||||
|
from src.content_labels import store_content_labels
|
||||||
|
|
||||||
|
|
||||||
def _store_last_post_id(base_dir: str, nickname: str, domain: str,
|
def _store_last_post_id(base_dir: str, nickname: str, domain: str,
|
||||||
|
|
@ -2838,6 +2839,15 @@ def _inbox_after_initial(server, inbox_start_time,
|
||||||
debug)
|
debug)
|
||||||
inbox_start_time = time.time()
|
inbox_start_time = time.time()
|
||||||
|
|
||||||
|
store_content_labels(base_dir, handle_name, domain,
|
||||||
|
http_prefix, domain_full,
|
||||||
|
post_json_object, session)
|
||||||
|
fitness_performance(inbox_start_time,
|
||||||
|
server.fitness,
|
||||||
|
'INBOX', 'store_content_labels',
|
||||||
|
debug)
|
||||||
|
inbox_start_time = time.time()
|
||||||
|
|
||||||
# send the post out to group members
|
# send the post out to group members
|
||||||
if is_group:
|
if is_group:
|
||||||
_send_to_group_members(server,
|
_send_to_group_members(server,
|
||||||
|
|
|
||||||
49
src/maps.py
49
src/maps.py
|
|
@ -1099,6 +1099,55 @@ def add_tag_map_links(tag_maps_dir: str, tag_name: str,
|
||||||
'EX: error writing tag map ' + tag_map_filename)
|
'EX: error writing tag map ' + tag_map_filename)
|
||||||
|
|
||||||
|
|
||||||
|
def add_label_map_links(labels_maps_dir: str, label: str,
|
||||||
|
map_links: [], published: str, post_url: str) -> None:
|
||||||
|
"""Appends to a hashtag file containing map links
|
||||||
|
This is used to show a map for a particular hashtag
|
||||||
|
"""
|
||||||
|
labels_map_filename: str = labels_maps_dir + '/' + label + '.txt'
|
||||||
|
post_url = post_url.replace('#', '/')
|
||||||
|
|
||||||
|
# read the existing map links
|
||||||
|
existing_map_links: list[str] = []
|
||||||
|
if is_a_file(labels_map_filename):
|
||||||
|
existing_map_links_str = \
|
||||||
|
load_string(labels_map_filename,
|
||||||
|
'EX: error reading label map ' + labels_map_filename)
|
||||||
|
if existing_map_links_str:
|
||||||
|
existing_map_links = existing_map_links_str.split('\n')
|
||||||
|
|
||||||
|
# combine map links with the existing list
|
||||||
|
secs_since_epoch: int = \
|
||||||
|
int((date_from_string_format(published, ['%Y-%m-%dT%H:%M:%S%z']) -
|
||||||
|
date_epoch()).total_seconds())
|
||||||
|
links_changed: bool = False
|
||||||
|
for link in map_links:
|
||||||
|
line = str(secs_since_epoch) + ' ' + link + ' ' + post_url
|
||||||
|
if line in existing_map_links:
|
||||||
|
continue
|
||||||
|
links_changed = True
|
||||||
|
existing_map_links = [line] + existing_map_links
|
||||||
|
if not links_changed:
|
||||||
|
return
|
||||||
|
|
||||||
|
# sort the list of map links
|
||||||
|
existing_map_links.sort(reverse=True)
|
||||||
|
map_links_str: str = ''
|
||||||
|
ctr: int = 0
|
||||||
|
for link in existing_map_links:
|
||||||
|
if not link:
|
||||||
|
continue
|
||||||
|
map_links_str += link + '\n'
|
||||||
|
ctr += 1
|
||||||
|
# don't allow the list to grow indefinitely
|
||||||
|
if ctr >= 2000:
|
||||||
|
break
|
||||||
|
|
||||||
|
# save the tag
|
||||||
|
save_string(map_links_str, labels_map_filename,
|
||||||
|
'EX: error writing label map ' + labels_map_filename)
|
||||||
|
|
||||||
|
|
||||||
def _gpx_location(latitude: float, longitude: float, post_id: str) -> str:
|
def _gpx_location(latitude: float, longitude: float, post_id: str) -> str:
|
||||||
"""Returns a gpx waypoint
|
"""Returns a gpx waypoint
|
||||||
"""
|
"""
|
||||||
|
|
|
||||||
135
src/utils.py
135
src/utils.py
|
|
@ -46,6 +46,17 @@ VALID_HASHTAG_CHARS = \
|
||||||
'ŔŕŘřẞߌśŜŝŞşŠšȘșŤťŢţÞþȚțÜüÙùÚúÛûŰűŨũŲųŮůŪū' +
|
'ŔŕŘřẞߌśŜŝŞşŠšȘșŤťŢţÞþȚțÜüÙùÚúÛûŰűŨũŲųŮůŪū' +
|
||||||
'ŴŵÝýŸÿŶŷŹźŽžŻż')
|
'ŴŵÝýŸÿŶŷŹźŽžŻż')
|
||||||
|
|
||||||
|
VALID_LABEL_CHARS = \
|
||||||
|
set(' _0123456789' +
|
||||||
|
'abcdefghijklmnopqrstuvwxyz' +
|
||||||
|
'ABCDEFGHIJKLMNOPQRSTUVWXYZ' +
|
||||||
|
'¡¿ÄäÀàÁáÂâÃãÅåǍǎĄąĂăÆæĀā' +
|
||||||
|
'ÇçĆćĈĉČčĎđĐďðÈèÉéÊêËëĚěĘęĖėĒē' +
|
||||||
|
'ĜĝĢģĞğĤĥÌìÍíÎîÏïıĪīĮįĴĵĶķ' +
|
||||||
|
'ĹĺĻļŁłĽľĿŀÑñŃńŇňŅņÖöÒòÓóÔôÕõŐőØøŒœ' +
|
||||||
|
'ŔŕŘřẞߌśŜŝŞşŠšȘșŤťŢţÞþȚțÜüÙùÚúÛûŰűŨũŲųŮůŪū' +
|
||||||
|
'ŴŵÝýŸÿŶŷŹźŽžŻż')
|
||||||
|
|
||||||
# posts containing these strings will always get screened out,
|
# posts containing these strings will always get screened out,
|
||||||
# both incoming and outgoing.
|
# both incoming and outgoing.
|
||||||
# Could include dubious clacks or admin dogwhistles
|
# Could include dubious clacks or admin dogwhistles
|
||||||
|
|
@ -2049,7 +2060,7 @@ def _remove_post_id_from_tag_index(tag_index_filename: str,
|
||||||
newlines += file_line
|
newlines += file_line
|
||||||
if not newlines.strip():
|
if not newlines.strip():
|
||||||
# if there are no lines then remove the hashtag file
|
# if there are no lines then remove the hashtag file
|
||||||
ex_text = 'EX: _delete_hashtags_on_post ' + \
|
ex_text = 'EX: _remove_post_id_from_tag_index ' + \
|
||||||
'unable to delete tag index ' + str(tag_index_filename)
|
'unable to delete tag index ' + str(tag_index_filename)
|
||||||
erase_file(tag_index_filename, ex_text)
|
erase_file(tag_index_filename, ex_text)
|
||||||
else:
|
else:
|
||||||
|
|
@ -2059,6 +2070,34 @@ def _remove_post_id_from_tag_index(tag_index_filename: str,
|
||||||
tag_index_filename)
|
tag_index_filename)
|
||||||
|
|
||||||
|
|
||||||
|
def _remove_post_id_from_label_index(labels_index_filename: str,
|
||||||
|
post_id: str) -> None:
|
||||||
|
"""Remove post_id from the label index file
|
||||||
|
"""
|
||||||
|
lines: list[str] = \
|
||||||
|
load_list(labels_index_filename,
|
||||||
|
'EX: _remove_post_id_from_label_index unable to read ' +
|
||||||
|
labels_index_filename)
|
||||||
|
if not lines:
|
||||||
|
return
|
||||||
|
newlines: str = ''
|
||||||
|
for file_line in lines:
|
||||||
|
if post_id in file_line:
|
||||||
|
# skip over the deleted post
|
||||||
|
continue
|
||||||
|
newlines += file_line
|
||||||
|
if not newlines.strip():
|
||||||
|
# if there are no lines then remove the hashtag file
|
||||||
|
ex_text = 'EX: _remove_post_id_from_label_index ' + \
|
||||||
|
'unable to delete label index ' + str(labels_index_filename)
|
||||||
|
erase_file(labels_index_filename, ex_text)
|
||||||
|
else:
|
||||||
|
# write the new label index without the given post in it
|
||||||
|
save_string(newlines, labels_index_filename,
|
||||||
|
'EX: _remove_post_id_from_label_index unable to write ' +
|
||||||
|
labels_index_filename)
|
||||||
|
|
||||||
|
|
||||||
def _delete_hashtags_on_post(base_dir: str, post_json_object: {}) -> None:
|
def _delete_hashtags_on_post(base_dir: str, post_json_object: {}) -> None:
|
||||||
"""Removes hashtags when a post is deleted
|
"""Removes hashtags when a post is deleted
|
||||||
"""
|
"""
|
||||||
|
|
@ -2096,6 +2135,84 @@ def _delete_hashtags_on_post(base_dir: str, post_json_object: {}) -> None:
|
||||||
_remove_post_id_from_tag_index(tag_index_filename, post_id)
|
_remove_post_id_from_tag_index(tag_index_filename, post_id)
|
||||||
|
|
||||||
|
|
||||||
|
def _delete_labels_on_post(base_dir: str, post_json_object: {}) -> None:
|
||||||
|
"""Removes labels when a post is deleted
|
||||||
|
"""
|
||||||
|
tags_list: list[dict] = []
|
||||||
|
obj: dict = post_json_object
|
||||||
|
if has_object_dict(post_json_object):
|
||||||
|
obj = post_json_object['object']
|
||||||
|
if 'tag' in obj:
|
||||||
|
if isinstance(obj['tag'], list):
|
||||||
|
tags_list = obj['tag']
|
||||||
|
if not tags_list:
|
||||||
|
return
|
||||||
|
if not obj.get('id'):
|
||||||
|
return
|
||||||
|
labels: list[str] = []
|
||||||
|
for tag_dict in tags_list:
|
||||||
|
if not isinstance(tag_dict, dict):
|
||||||
|
continue
|
||||||
|
if 'type' not in tag_dict or 'name' not in tag_dict:
|
||||||
|
continue
|
||||||
|
if 'value' not in tag_dict and 'href' not in tag_dict:
|
||||||
|
continue
|
||||||
|
if not isinstance(tag_dict['type'], str):
|
||||||
|
continue
|
||||||
|
if not isinstance(tag_dict['name'], str):
|
||||||
|
continue
|
||||||
|
if tag_dict['type'] == 'PropertyValue' and \
|
||||||
|
'value' in tag_dict:
|
||||||
|
if not isinstance(tag_dict['value'], str):
|
||||||
|
continue
|
||||||
|
if tag_dict['name'] == 'Labels':
|
||||||
|
if ',' in tag_dict['value']:
|
||||||
|
labels_list = tag_dict['value'].split(',')
|
||||||
|
else:
|
||||||
|
labels_list = tag_dict['value'].split('/')
|
||||||
|
for label_str in labels_list:
|
||||||
|
label_str = label_str.strip()
|
||||||
|
label_str = remove_html(label_str)
|
||||||
|
if label_str:
|
||||||
|
if label_str not in labels:
|
||||||
|
labels.append(label_str)
|
||||||
|
elif tag_dict['name'] == 'Label':
|
||||||
|
label_str = tag_dict['value'].strip()
|
||||||
|
label_str = remove_html(label_str)
|
||||||
|
if label_str:
|
||||||
|
if 'href' in tag_dict:
|
||||||
|
if isinstance(tag_dict['href'], str):
|
||||||
|
if resembles_url(tag_dict['href']):
|
||||||
|
label_str += '###' + tag_dict['href']
|
||||||
|
if label_str not in labels:
|
||||||
|
labels.append(label_str)
|
||||||
|
elif tag_dict['type'] == 'Label':
|
||||||
|
label_str = remove_html(tag_dict['name'])
|
||||||
|
if label_str:
|
||||||
|
if 'href' in tag_dict:
|
||||||
|
if isinstance(tag_dict['href'], str):
|
||||||
|
if resembles_url(tag_dict['href']):
|
||||||
|
label_str += '###' + tag_dict['href']
|
||||||
|
if label_str not in labels:
|
||||||
|
labels.append(label_str)
|
||||||
|
if not labels:
|
||||||
|
return
|
||||||
|
|
||||||
|
# get the id of the post
|
||||||
|
post_id: str = remove_id_ending(post_json_object['object']['id'])
|
||||||
|
for label in labels:
|
||||||
|
# find the index file for this label
|
||||||
|
labels_map_filename: str = \
|
||||||
|
base_dir + '/labelsmaps/' + label + '.txt'
|
||||||
|
if is_a_file(labels_map_filename):
|
||||||
|
_remove_post_id_from_label_index(labels_map_filename, post_id)
|
||||||
|
# find the index file for this tag
|
||||||
|
labels_index_filename: str = \
|
||||||
|
base_dir + '/labels/' + label + '.txt'
|
||||||
|
if is_a_file(labels_index_filename):
|
||||||
|
_remove_post_id_from_label_index(labels_index_filename, post_id)
|
||||||
|
|
||||||
|
|
||||||
def _delete_conversation_post(base_dir: str, nickname: str, domain: str,
|
def _delete_conversation_post(base_dir: str, nickname: str, domain: str,
|
||||||
post_json_object: {}) -> None:
|
post_json_object: {}) -> None:
|
||||||
"""Deletes a post from a conversation
|
"""Deletes a post from a conversation
|
||||||
|
|
@ -2429,9 +2546,10 @@ def delete_post(base_dir: str, http_prefix: str,
|
||||||
post_id: str = remove_id_ending(post_json_object['id'])
|
post_id: str = remove_id_ending(post_json_object['id'])
|
||||||
remove_moderation_post_from_index(base_dir, post_id, debug)
|
remove_moderation_post_from_index(base_dir, post_id, debug)
|
||||||
|
|
||||||
# remove any hashtags index entries
|
# remove any hashtags and labels index entries
|
||||||
if has_object:
|
if has_object:
|
||||||
_delete_hashtags_on_post(base_dir, post_json_object)
|
_delete_hashtags_on_post(base_dir, post_json_object)
|
||||||
|
_delete_labels_on_post(base_dir, post_json_object)
|
||||||
|
|
||||||
# remove any replies
|
# remove any replies
|
||||||
_delete_post_remove_replies(base_dir, nickname, domain,
|
_delete_post_remove_replies(base_dir, nickname, domain,
|
||||||
|
|
@ -3314,6 +3432,19 @@ def valid_hash_tag(hashtag: str) -> bool:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def valid_content_label(label: str) -> bool:
|
||||||
|
"""Returns true if the give content label contains valid characters
|
||||||
|
"""
|
||||||
|
# long labels are not valid
|
||||||
|
if len(label) >= 32:
|
||||||
|
return False
|
||||||
|
if set(label).issubset(VALID_LABEL_CHARS):
|
||||||
|
return True
|
||||||
|
if _is_valid_language(label):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def load_bold_reading(base_dir: str) -> {}:
|
def load_bold_reading(base_dir: str) -> {}:
|
||||||
"""Returns a dictionary containing the bold reading status for each account
|
"""Returns a dictionary containing the bold reading status for each account
|
||||||
"""
|
"""
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue