import json import html2text import re import sqlite3 from pathlib import Path from mlx_lm.utils import load # type: ignore from mlx_lm.generate import generate # type: ignore model, tokenizer = load("mlx-community/SmolLM3-3B-8bit") system_prompt = """/no_think You are a JSON generator. You are a JSON generator. Extract ALL people, places, and topics from the provided input. Please use structured JSON for the output: ```json { "people": [ "string", "string", etc. ], "places": [ "string", "string", etc. ], "topics": [ "string", "string", etc. ], } ``` Return only JSON inside these markers: { ... } Important: Your entire output must be a single valid JSON object. Do not add comments, explanations, or text outside the JSON braces. If you cannot comply, output {}. """ def init_database(db_path="posts_processed.db"): """Initialize SQLite database with posts table""" conn = sqlite3.connect(db_path) cursor = conn.cursor() # Create table if it doesn't exist cursor.execute(''' CREATE TABLE IF NOT EXISTS processed_posts ( post_id TEXT PRIMARY KEY, json_data TEXT NOT NULL, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP ) ''') conn.commit() return conn def post_exists(conn, post_id): """Check if a post_id already exists in the database""" cursor = conn.cursor() cursor.execute('SELECT 1 FROM processed_posts WHERE post_id = ?', (post_id,)) return cursor.fetchone() is not None def save_post(conn, post_id, json_data): """Save processed post to database""" cursor = conn.cursor() cursor.execute(''' INSERT INTO processed_posts (post_id, json_data) VALUES (?, ?) ''', (post_id, json.dumps(json_data))) conn.commit() def extract_json_from_response(response): """Extract JSON from the model response between markers""" try: # Look for content between and markers import re json_match = re.search(r'\s*(.*?)\s*', response, re.DOTALL) if json_match: json_str = json_match.group(1) return json.loads(json_str) else: # If no markers found, try to parse the entire response as JSON # First, try to find a JSON object in the response json_match = re.search(r'\{.*\}', response, re.DOTALL) if json_match: return json.loads(json_match.group(0)) except json.JSONDecodeError as e: print(f"Failed to parse JSON: {e}") except Exception as e: print(f"Error extracting JSON: {e}") return {} def load_all_posts(posts_dir="posts"): """Load all posts from all JSON files into a single list""" all_posts = [] posts_path = Path(posts_dir) for json_file in posts_path.glob("*.json"): try: with open(json_file, 'r', encoding='utf-8') as f: posts = json.load(f) all_posts.extend(posts) except Exception as e: print(f"Error reading {json_file}: {e}") return all_posts def main(): # Initialize database conn = init_database() # Load posts posts = load_all_posts() total_posts = len(posts) processed_count = 0 skipped_count = 0 for i, post in enumerate(posts, 1): post_id = post['post_id'] # Skip if post already exists in database if post_exists(conn, post_id): skipped_count += 1 print(f"\rProgress: {i}/{total_posts} | Processed: {processed_count} | Skipped: {skipped_count}", end="", flush=True) continue # Extract and clean text html = post['post_html'] if html is None: html = "" text2 = html2text.html2text(html) cleaned = re.sub(r'https?://\S+|www\.\S+', '', text2) prompt = f""" {cleaned} """ messages = [ { "role": "system", "content": system_prompt }, { "role": "user", "content": prompt } ] prompt = tokenizer.apply_chat_template(messages, add_generation_prompt=True) response = generate( model, tokenizer, prompt=prompt, verbose=False, max_tokens=200 ) # Extract JSON from response and save to database extracted_json = extract_json_from_response(response) if extracted_json: save_post(conn, post_id, extracted_json) processed_count += 1 print(f"\rProgress: {i}/{total_posts} | Processed: {processed_count} | Skipped: {skipped_count}", end="", flush=True) # Close database connection conn.close() print(f"\n✓ Complete: Processed {processed_count} new posts, skipped {skipped_count} existing posts") if __name__ == "__main__": main()