Repository navigation
Expand file tree
/
Copy pathapp.py
More file actions
152 lines (117 loc) · 4.62 KB
/
Copy pathapp.py
File metadata and controls
152 lines (117 loc) · 4.62 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
import os
import secrets
import pdfplumber
from PyPDF2 import PdfReader
from flask import Flask, render_template, request, redirect, url_for, session, flash
from werkzeug.utils import secure_filename
from syllabus_processor import extract_topics
from scheduler import generate_schedule
from video_recommender import recommend_videos
app = Flask(__name__)
app.secret_key = os.getenv('SECRET_KEY') or secrets.token_hex(16)
app.config['MAX_CONTENT_LENGTH'] = 16 * 1024 * 1024
UPLOAD_FOLDER = 'uploads'
ALLOWED_EXTENSIONS = {'pdf', 'txt'}
app.config['UPLOAD_FOLDER'] = UPLOAD_FOLDER
os.makedirs(UPLOAD_FOLDER, exist_ok=True)
def allowed_file(filename):
return '.' in filename and filename.rsplit('.', 1)[1].lower() in ALLOWED_EXTENSIONS
def pull_text_from_pdf(filepath):
"""
Try pdfplumber first; if that yields no text, fall back to PyPDF2.
"""
text_chunks = []
# First pass: pdfplumber
try:
with pdfplumber.open(filepath) as pdf:
for page in pdf.pages:
txt = page.extract_text()
if txt:
text_chunks.append(txt)
except Exception as e:
app.logger.debug(f"pdfplumber error: {e}")
full_text = "\n".join(text_chunks).strip()
if full_text:
app.logger.debug(f"PDF text length via pdfplumber: {len(full_text)}")
return full_text
# Fallback to PyPDF2
try:
reader = PdfReader(filepath)
fallback = []
for page in reader.pages:
t = page.extract_text()
if t:
fallback.append(t)
result = "\n".join(fallback).strip()
app.logger.debug(f"PDF text length via PyPDF2: {len(result) if result else 0}")
return result if result else None
except Exception as err:
flash(f"PDF parse error: {err}", 'danger')
return None
def pull_text_from_txt(filepath):
"""
Try reading the TXT file with multiple encodings and log a snippet.
"""
for enc in ('utf-8', 'latin-1', 'utf-16'):
try:
with open(filepath, 'r', encoding=enc) as f:
data = f.read()
if data:
snippet = data[:200].replace('\n', ' ').replace('\r', '')
app.logger.debug(f"[TXT-{enc}] snippet: {snippet!r}")
return data.strip()
except Exception as e:
app.logger.debug(f"[TXT-{enc}] read error: {e}")
continue
flash("Unable to read text file (tried UTF-8, Latin-1, UTF-16).", 'danger')
return None
@app.route('/')
def index():
return render_template('index.html')
@app.route('/upload', methods=['GET', 'POST'])
def upload():
if request.method == 'POST':
file = request.files.get('file')
if not file or not allowed_file(file.filename):
flash("Only PDF or TXT files are allowed.", 'danger')
return redirect(request.url)
filename = secure_filename(file.filename)
path = os.path.join(app.config['UPLOAD_FOLDER'], filename)
file.save(path)
# Extract text based on extension
if filename.lower().endswith('.pdf'):
text = pull_text_from_pdf(path)
else:
text = pull_text_from_txt(path)
# Clean up the upload immediately
os.remove(path)
# Log and handle missing text
if not text:
app.logger.debug("No text extracted from file.")
return redirect(request.url)
app.logger.debug(f"Total text length: {len(text)}")
# Extract topics
topics = extract_topics(text)
app.logger.debug(f"Topics extracted: {topics}")
if not topics:
flash("No topics found in that file.", 'warning')
return redirect(request.url)
session['topics'] = topics
return redirect(url_for('dashboard'))
return render_template('upload.html')
@app.route('/dashboard')
def dashboard():
topics = session.get('topics', [])
if not topics:
flash("Please upload a syllabus first.", 'info')
return redirect(url_for('upload'))
video_links = recommend_videos(topics)
return render_template('dashboard.html', topics=topics, video_links=video_links)
@app.route('/analyze_and_plan', methods=['POST'])
def analyze_and_plan():
topics = session.get('topics', [])
schedule = generate_schedule(topics) if topics else None
return render_template('result.html', schedule=schedule)
if __name__ == '__main__':
# Enable debug logging
app.run(debug=True)