Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/deploy_staging.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@ name: Deploy
on:
push:
branches:
- web
- staging

jobs:
pull:
Expand Down
864 changes: 864 additions & 0 deletions client.bpmn

Large diffs are not rendered by default.

12 changes: 10 additions & 2 deletions exorde/arguments.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@
import re
import logging
import os
import sys


def write_env(email, password, username, http_proxy=""):
Expand Down Expand Up @@ -56,6 +57,11 @@ def clear_env():
logging.info("clear_env: .env file cleared.")


def hide_stdout():
# Redirect stdout to os.devnull
sys.stdout = open(os.devnull, "w")


def setup_arguments() -> argparse.Namespace:
def batch_size_type(value):
ivalue = int(value)
Expand Down Expand Up @@ -119,7 +125,7 @@ def validate_quota_spec(quota_spec: str) -> dict:
action="append", # allow reuse of the option in the same run
help="quota a domain per 24h (domain=amount)",
)

parser.add_argument("--noout", action="store_true", help="Hide stdout")
parser.add_argument(
"-ntfy",
"--ntfy",
Expand Down Expand Up @@ -164,6 +170,9 @@ def parse_list(s):
)
args = parser.parse_args()

if args.noout:
hide_stdout()

# Check that either all or none of Twitter arguments are provided
args_list = [
args.twitter_username,
Expand Down Expand Up @@ -197,7 +206,6 @@ def parse_list(s):
"[Init] No login arguments detected: using login-less scraping"
)
clear_env()

command_line_arguments: argparse.Namespace = parser.parse_args()
if len(command_line_arguments.notify_at) == 0:
command_line_arguments.notify_at = [12, 19]
Expand Down
20 changes: 18 additions & 2 deletions exorde/brain.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,8 +2,9 @@
import json
import logging
import argparse
from exorde.get_keywords import choose_keyword
from exorde.get_keywords import choose_keyword, get_keywords
from exorde.module_loader import get_scraping_module
from exorde.expiration_calc import retrieve_and_calculate_word_freq
import aiohttp
import datetime
from typing import Union, Callable
Expand Down Expand Up @@ -291,19 +292,34 @@ async def think(
if remaining_iterations_looping <= 0:
break

keyword: str = await choose_keyword(
(keyword, alg) = await choose_keyword(
module.__name__, ponderation, websocket_send, intent_id
)
"""
These parameters are module parameters defined trough configuration files
"""
generic_modules_parameters: dict[
str, Union[int, str, bool, dict]
] = ponderation.generic_modules_parameters
specific_parameters: dict[
str, Union[int, str, bool, dict]
] = ponderation.specific_modules_parameters.get(choosen_module_path, {})
# specific parameters can contain
# - max_oldness_seconds
# - pick_default_keyword_weight
# - maximum_items_to_collect
# - and others
parameters: dict[str, Union[int, str, bool, dict]] = {
"url_parameters": {"keyword": keyword},
"keyword": keyword,
}
parameters.update(generic_modules_parameters)
parameters.update(specific_parameters)
"""
if we are using the 'old' alg we should overwrite the 'max_oldness_seconds'
parameters by the one retrieved from the expiration_calculator
"""
if alg == 'old':
word_freq = await retrieve_and_calculate_word_freq()
parameters['max_oldness_seconds'] = word_freq.get(keyword, 42000)
return (module, parameters, domain)
51 changes: 51 additions & 0 deletions exorde/cache.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@

import functools
import time

def cached(duration):
"""
A decorator that caches the result of a function for a specified duration.

Usage:
@cached(duration_in_seconds)
def my_function(args):
...

Parameters:
- duration (int): The duration for which the result of the function should
be cached in seconds.

Returns:
- function: A decorated function that caches its results.

Example:
@cached(5 * 60) # Cache results for 5 minutes
def add(a, b):
return a + b

# The first call to add(2, 3) will calculate the result (5) and cache it.
# Subsequent calls within the next 5 minutes will return the cached result.
result = add(2, 3)
"""
def decorator(func):
cache = {}

@functools.wraps(func)
async def wrapper(*args, **kwargs):
key = (args, frozenset(kwargs.items()))

# Check if the result is cached and not expired
if key in cache and time.time() - cache[key]['timestamp'] < duration:
return cache[key]['result']

# Calculate and cache the result
result = await func(*args, **kwargs)
cache[key] = {
'result': result,
'timestamp': time.time(),
}
return result

return wrapper

return decorator
30 changes: 30 additions & 0 deletions exorde/expiration_calc.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
"""
Calculate maximum expiration date of collected items base on their appearence
frequency in the keyword list

https://raw.githubusercontent.com/exorde-labs/TestnetProtocol/main/targets/keywords.txt
"""
import aiohttp
import asyncio
from collections import Counter
from exorde.get_keywords import get_keywords
from exorde.cache import cached

def calculate_word_freq(text_content):
if text_content is not None:
# Split the content into lines and remove trailing '\r'
lines = [line.strip('\r').lower() for line in text_content]

# Calculate the frequency of each word using Counter
word_freq = Counter(lines)

return dict(word_freq)
else:
return {}


@cached(6 * 60) # might collide with get_keywords cache so we delay it a lil # todo improve using a priority queue
async def retrieve_and_calculate_word_freq():
text_content = await get_keywords()
word_freq = calculate_word_freq(text_content)
return word_freq
Loading