Downloads · 30 days
0
pugfal67/dpbh
dpbh is a machine learning model from pugfal67. Use it for the machine learning task on the model card, and read the license before you ship it in a product. The card lists the license as apache-2.0.
import aiohttp import asyncio from bs4 import BeautifulSoup from transformers import pipeline from flask import Flask, request, jsonify
Downloads · 30 days
0
Access
Public
Updated Jan 24, 2024
Repo size
—
Likes
0
Public
Click a slice to open those files.
.md3.3 KB · 69%
From the Hugging Face model README
import aiohttp import asyncio from bs4 import BeautifulSoup from transformers import pipeline from flask import Flask, request, jsonify
app = Flask(name)
class EcommerceDarkPatternDetector: def init(self): self.fake_timer_keywords = ["hurry", "limited time", "almost gone", "only a few left"] self.hidden_cost_keywords = ["shipping", "tax", "additional charges", "fees"] self.learning_data = [] self.nlp_model = pipeline("feature-extraction", model="bert-base-uncased", tokenizer="bert-base-uncased")
async def async_scrape_website(self, url):
async with aiohttp.ClientSession() as session:
async with session.get(url) as response:
return await response.text()
async def detect_fake_urgency(self, product_description):
return any(re.search(keyword, product_description, flags=re.IGNORECASE) for keyword in self.fake_timer_keywords)
async def identify_hidden_costs(self, checkout_page):
return any(re.search(keyword, checkout_page, flags=re.IGNORECASE) for keyword in self.hidden_cost_keywords)
async def analyze_product_information(self, product_details):
vectors = self.nlp_model([product_details] + self.learning_data)
similarity = (vectors[0] @ vectors[1:].T).mean()
return similarity.item()
async def check_website(self, url):
try:
# Implement caching mechanism here to avoid repeated requests
website_content = await self.async_scrape_website(url)
if not website_content:
return None
soup = BeautifulSoup(website_content, 'lxml') # Use lxml for faster parsing
product_description = soup.find('meta', {'name': 'description'})['content'] if soup.find('meta', {'name': 'description'}) else ""
checkout_page = soup.find('div', {'class': 'checkout-page'}).text if soup.find('div', {'class': 'checkout-page'}) else ""
fake_urgency_detected = await self.detect_fake_urgency(product_description)
hidden_costs_identified = await self.identify_hidden_costs(checkout_page)
# Analyze product information and update learning data
product_similarity = await self.analyze_product_information(product_description)
self.learning_data.append(product_description)
return {
'fake_urgency_detected': fake_urgency_detected,
'hidden_costs_identified': hidden_costs_identified,
'product_similarity': product_similarity
}
except Exception as e:
print(f"Error while checking website: {e}")
return None
ecommerce_detector = EcommerceDarkPatternDetector()
@app.route('/detect-dark-patterns', methods=['POST']) def detect_dark_patterns(): data = request.get_json() website_url = data.get('url')
if not website_url:
return jsonify({"error": "Missing 'url' parameter"}), 400
loop = asyncio.get_event_loop()
result = loop.run_until_complete(ecommerce_detector.check_website(website_url))
if result:
return jsonify(result)
else:
return jsonify({"error": "Failed to fetch or analyze website content"}), 500
if name == "main": app.run(debug=True)