1use lopdf::{Document, Object};
7use regex::Regex;
8use std::path::Path;
9use thiserror::Error;
10
11#[derive(Error, Debug)]
13pub enum ParserError {
14 #[error("PDF parsing error: {0}")]
16 Pdf(#[from] lopdf::Error),
17 #[error("IO error: {0}")]
19 Io(#[from] std::io::Error),
20 #[error("Failed to extract text from PDF")]
22 ExtractionFailed,
23}
24
25#[derive(Debug, Clone)]
27pub struct Lesson {
28 pub subject: String,
30 pub room: String,
32 pub teacher: String,
34 pub class_code: String,
36 pub day_index: usize,
38 pub period_index: usize,
40}
41
42#[derive(Debug, Clone)]
44pub struct Week {
45 pub lessons: Vec<Lesson>,
47 pub week_name: String,
49 pub student_name: Option<String>,
51 pub form: Option<String>,
53}
54
55#[derive(Debug, Clone)]
57struct TextItem {
58 x: f64,
59 y: f64,
60 text: String,
61}
62
63pub fn parse_pdf(path: &Path) -> Result<Vec<Week>, ParserError> {
99 let doc = Document::load(path)?;
100 let mut weeks = Vec::new();
101
102 for (page_num, page_id) in doc.get_pages() {
103 let text_items = extract_text_from_page(&doc, page_id)?;
104 if text_items.is_empty() {
105 continue;
106 }
107
108 let page_weeks = process_page_text(text_items, page_num);
109 weeks.extend(page_weeks);
110 }
111
112 Ok(weeks)
113}
114
115fn extract_text_from_page(
116 doc: &Document,
117 page_id: (u32, u16),
118) -> Result<Vec<TextItem>, ParserError> {
119 let content_bytes = doc.get_page_content(page_id);
120 let content = lopdf::content::Content::decode(&content_bytes)?;
121 let mut text_items = Vec::new();
122
123 let mut current_x = 0.0;
124 let mut current_y = 0.0;
125
126 for operation in content.operations.iter() {
127 match operation.operator.as_str() {
128 "BT" => {
129 current_x = 0.0;
130 current_y = 0.0;
131 }
132 "Tm" => {
133 if operation.operands.len() == 6 {
134 if let (Ok(e), Ok(f)) = (
135 operation.operands[4].as_float(),
136 operation.operands[5].as_float(),
137 ) {
138 current_x = e as f64;
139 current_y = f as f64;
140 }
141 }
142 }
143 "Td" | "TD" => {
144 if operation.operands.len() == 2 {
145 if let (Ok(tx), Ok(ty)) = (
146 operation.operands[0].as_float(),
147 operation.operands[1].as_float(),
148 ) {
149 current_x += tx as f64;
150 current_y += ty as f64;
151 }
152 }
153 }
154 "Tj" => {
155 if let Some(text) = decode_text_object(&operation.operands[0]) {
156 text_items.push(TextItem {
157 x: current_x,
158 y: current_y,
159 text: decode_bromcom_text(&text),
160 });
161 }
162 }
163 "TJ" => {
164 if let Ok(arr) = operation.operands[0].as_array() {
165 let mut full_text = String::new();
166 for item in arr {
167 if let Some(text) = decode_text_object(item) {
168 full_text.push_str(&text);
169 }
170 }
171 text_items.push(TextItem {
172 x: current_x,
173 y: current_y,
174 text: decode_bromcom_text(&full_text),
175 });
176 }
177 }
178 _ => {}
179 }
180 }
181
182 Ok(text_items)
183}
184
185fn decode_text_object(obj: &Object) -> Option<String> {
186 match obj {
187 Object::String(bytes, _) => String::from_utf8(bytes.clone()).ok(),
188 _ => None,
189 }
190}
191
192fn decode_bromcom_text(text: &str) -> String {
193 text.chars()
194 .filter(|&c| c != '\0')
195 .map(|c| {
196 let code = c as u8;
197 let new_code = code.wrapping_add(29);
198 new_code as char
199 })
200 .collect()
201}
202
203fn process_page_text(items: Vec<TextItem>, _page_num: u32) -> Vec<Week> {
204 let mut weeks = Vec::new();
205
206 let week_regex = Regex::new(r"Week\s+(\d+)").unwrap();
207
208 let mut week_headers: Vec<(&TextItem, u32)> = items
210 .iter()
211 .filter_map(|i| {
212 week_regex
213 .captures(&i.text)
214 .map(|cap| (i, cap[1].parse::<u32>().unwrap_or(0)))
215 })
216 .collect();
217
218 week_headers.sort_by_key(|k| k.1);
220
221 if week_headers.is_empty() {
222 return weeks;
223 }
224
225 let y_increases_down = if week_headers.len() > 1 {
228 week_headers[1].0.y > week_headers[0].0.y
229 } else {
230 let header_y = week_headers[0].0.y;
232 let items_below_y_down = items.iter().filter(|i| i.y > header_y).count();
233 let items_below_y_up = items.iter().filter(|i| i.y < header_y).count();
234 items_below_y_down > items_below_y_up
235 };
236
237 for (i, (header, _week_num)) in week_headers.iter().enumerate() {
238 let start_y = header.y;
239 let end_y = if i + 1 < week_headers.len() {
240 week_headers[i + 1].0.y
241 } else if y_increases_down {
242 f64::MAX
243 } else {
244 0.0
245 };
246
247 let week_items: Vec<&TextItem> = items
264 .iter()
265 .filter(|item| {
266 if y_increases_down {
267 item.y >= start_y && item.y < end_y
268 } else {
269 item.y <= start_y && item.y > end_y
270 }
271 })
272 .collect();
273
274 let week_name = if let Some(mat) = week_regex.find(&header.text) {
276 mat.as_str().to_string()
277 } else {
278 "Unknown Week".to_string()
279 };
280
281 let lessons = parse_week_items(&week_items);
282
283 let (student_name, form) = extract_student_info(&week_items);
285
286 if !lessons.is_empty() {
287 weeks.push(Week {
288 lessons,
289 week_name,
290 student_name,
291 form,
292 });
293 }
294 }
295
296 weeks
297}
298
299fn parse_week_items(items: &[&TextItem]) -> Vec<Lesson> {
300 let mut lessons = Vec::new();
301
302 let days = ["Monday", "Tuesday", "Wednesday", "Thursday", "Friday"];
304 let mut day_cols: Vec<(usize, f64)> = Vec::new(); for (i, day) in days.iter().enumerate() {
307 if let Some(header) = items.iter().find(|item| {
308 item.text.trim().eq_ignore_ascii_case(day)
309 || item.text.to_lowercase().contains(&day.to_lowercase())
310 }) {
311 day_cols.push((i, header.x));
312 }
314 }
315
316 if day_cols.is_empty() {
317 return lessons;
322 }
323
324 let marker_map = [
329 ("PD", 0),
330 ("Reg", 0),
331 ("L1", 1),
332 ("1", 1),
333 ("L2", 2),
334 ("2", 2),
335 ("L3", 3),
336 ("3", 3),
337 ("L4", 4),
338 ("4", 4),
339 ("L4/", 4),
340 ("L5", 5),
341 ("5", 5),
342 ];
343
344 let mut period_rows: Vec<(usize, f64)> = Vec::new(); for (marker_text, period_idx) in marker_map.iter() {
347 let matching_items: Vec<&f64> = items
349 .iter()
350 .filter(|item| {
351 let text = item.text.trim();
352 text == *marker_text ||
353 (marker_text.len() == 2 && text.starts_with(marker_text))
355 })
356 .map(|item| &item.y)
357 .collect();
358
359 if !matching_items.is_empty() {
360 let avg_y: f64 =
362 matching_items.iter().cloned().sum::<f64>() / matching_items.len() as f64;
363 if !period_rows.iter().any(|(idx, _)| idx == period_idx) {
365 period_rows.push((*period_idx, avg_y));
366 }
367 }
368 }
369
370 let teacher_regex_filter = Regex::new(r"^(Mr|Ms|Mrs|Miss)\s+.*$").unwrap();
373 for (day_idx, day_x) in &day_cols {
374 for (period_idx, period_y) in &period_rows {
375 let main_y_tolerance = if *period_idx == 0 { 26.0 } else { 25.0 };
386 let main_items: Vec<&&TextItem> = items
387 .iter()
388 .filter(|item| {
389 (item.x - day_x).abs() < 45.0 &&
390 (item.y - period_y).abs() <= main_y_tolerance &&
391 !days.iter().any(|d| item.text.trim().eq_ignore_ascii_case(d)) &&
393 !marker_map.iter().any(|(m, _)| item.text.trim() == *m)
394 })
395 .collect();
396
397 let teacher_items: Vec<&&TextItem> = items
399 .iter()
400 .filter(|item| {
401 (item.x - day_x).abs() < 45.0 &&
402 item.y > *period_y && (item.y - period_y).abs() < 35.0 &&
404 teacher_regex_filter.is_match(item.text.trim())
405 })
406 .collect();
407
408 let mut cell_items: Vec<&&TextItem> = main_items;
410 cell_items.extend(teacher_items);
411
412 if !cell_items.is_empty() {
413 let lesson = parse_lesson_content(cell_items, *day_idx, *period_idx);
414 lessons.push(lesson);
415 }
416 }
417 }
418
419 lessons
420}
421
422fn parse_lesson_content(items: Vec<&&TextItem>, day_index: usize, period_index: usize) -> Lesson {
423 let mut sorted_items = items.clone();
425 sorted_items.sort_by(|a, b| {
426 a.y.partial_cmp(&b.y)
427 .unwrap_or(std::cmp::Ordering::Equal)
428 .then(a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal))
429 });
430
431 let mut subject_parts: Vec<String> = Vec::new();
432 let mut room = "Unknown".to_string();
433 let mut teacher = "Unknown".to_string();
434 let mut class_code = String::new();
435
436 let room_regex = Regex::new(r"^[A-Z]{2,3}\d+[A-Z]?$").unwrap(); let teacher_regex = Regex::new(r"^(Mr|Ms|Mrs|Miss)\s+.*$").unwrap();
438 let class_regex = Regex::new(r"^\d{1,2}[A-Z][A-Za-z0-9]*(?:/[A-Za-z0-9]+)?$").unwrap();
441 let days = ["Monday", "Tuesday", "Wednesday", "Thursday", "Friday"];
442
443 let location_indicators = ["DEFAULT", "DS"];
445
446 for item in sorted_items {
447 let text = item.text.trim();
448 if text.is_empty() {
449 continue;
450 }
451
452 if days.iter().any(|d| text.eq_ignore_ascii_case(d)) {
454 continue;
455 }
456
457 if location_indicators.contains(&text) {
459 continue;
460 }
461
462 if room_regex.is_match(text) && room == "Unknown" {
463 room = text.to_string();
465 } else if teacher_regex.is_match(text) {
466 teacher = text.to_string();
467 } else if class_regex.is_match(text) {
468 class_code = text.to_string();
469 } else {
470 subject_parts.push(text.to_string());
472 }
473 }
474
475 let subject = if subject_parts.is_empty() {
479 "Unknown".to_string()
480 } else {
481 abbreviate_subject(&subject_parts.join(" "))
482 };
483
484 Lesson {
485 subject,
486 room,
487 teacher,
488 class_code,
489 day_index,
490 period_index,
491 }
492}
493
494fn abbreviate_subject(subject: &str) -> String {
495 if subject
496 .split_whitespace()
497 .eq(["Personal", "Development", "Intervention"])
498 {
499 "PDI".to_string()
500 } else {
501 subject.to_string()
502 }
503}
504
505fn extract_student_info(items: &[&TextItem]) -> (Option<String>, Option<String>) {
506 let form_in_parens_regex = Regex::new(r"^(.+?)\s*\(([0-9]{2,4}[A-Z0-9]*)\)$").unwrap();
513 let form_code_regex = Regex::new(r"^[0-9]{2,4}[A-Z0-9]*$").unwrap(); if items.is_empty() {
516 return (None, None);
517 }
518
519 let mut sorted = items.to_vec();
521 sorted.sort_by(|a, b| {
522 a.y.partial_cmp(&b.y)
523 .unwrap_or(std::cmp::Ordering::Equal)
524 .then(a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal))
525 });
526
527 let mut student_name = None;
528 let mut form = None;
529
530 let excluded = [
532 "Week",
533 "Term",
534 "Monday",
535 "Tuesday",
536 "Wednesday",
537 "Thursday",
538 "Friday",
539 "Page",
540 "of",
541 "Personal",
542 "Development",
543 "Intervention",
544 ];
545
546 for item in sorted.iter().take(50) {
548 let text = item.text.trim();
549
550 if let Some(cap) = form_in_parens_regex.captures(text) {
551 let name = cap[1].trim();
552 if name.len() > 3 {
553 return (Some(name.to_string()), Some(cap[2].to_string()));
554 }
555 }
556 }
557
558 for (i, item) in sorted.iter().take(50).enumerate() {
560 let text = item.text.trim();
561
562 if text.is_empty() || excluded.iter().any(|&e| text.contains(e)) {
563 continue;
564 }
565
566 if form.is_none() && form_code_regex.is_match(text) {
568 form = Some(text.to_string());
569
570 if student_name.is_none() {
572 for j in (0..i).rev() {
573 let prev = sorted[j];
574 let prev_text = prev.text.trim();
575
576 if (prev.y - item.y).abs() < 5.0
578 && prev_text.len() > 3
579 && prev_text.contains(' ')
580 && !excluded.iter().any(|&e| prev_text.contains(e))
581 && !prev_text.starts_with("Mr")
582 && !prev_text.starts_with("Ms")
583 && !prev_text.starts_with("Mrs")
584 && !prev_text.starts_with("Miss")
585 {
586 student_name = Some(prev_text.to_string());
587 break;
588 }
589 }
590 }
591 }
592
593 if student_name.is_some() && form.is_some() {
595 break;
596 }
597 }
598
599 (student_name, form)
600}
601
602#[cfg(test)]
603mod tests {
604 use super::*;
605
606 fn make_item(x: f64, y: f64, text: &str) -> TextItem {
607 TextItem {
608 x,
609 y,
610 text: text.to_string(),
611 }
612 }
613
614 #[test]
615 fn parse_lesson_with_room_and_teacher() {
616 let src = [
617 make_item(100.0, 100.0, "Personal"),
618 make_item(100.0, 110.0, "Development"),
619 make_item(100.0, 120.0, "Intervention"),
620 make_item(100.0, 130.0, "HU9"),
621 make_item(100.0, 145.0, "Ms Test A"),
622 ];
623
624 let refs: Vec<&TextItem> = src.iter().collect();
625 let refsrefs: Vec<&&TextItem> = refs.iter().collect();
626
627 let lesson = parse_lesson_content(refsrefs, 0, 0);
628 assert_eq!(lesson.subject, "PDI");
629 assert_eq!(lesson.room, "HU9");
630 assert_eq!(lesson.teacher, "Ms Test A");
631 }
632
633 #[test]
634 fn parse_lesson_detects_classcode() {
635 let src = [
636 make_item(100.0, 200.0, "Science"),
637 make_item(100.0, 210.0, "8A1/Co"),
638 make_item(100.0, 220.0, "Mr Test B"),
639 ];
640
641 let refs: Vec<&TextItem> = src.iter().collect();
642 let refsrefs: Vec<&&TextItem> = refs.iter().collect();
643
644 let lesson = parse_lesson_content(refsrefs, 1, 2);
645 assert_eq!(lesson.subject, "Science");
646 assert_eq!(lesson.class_code, "8A1/Co");
647 assert_eq!(lesson.teacher, "Mr Test B");
648 }
649
650 #[test]
651 fn parse_week_keeps_room_after_wrapped_subject() {
652 let src = [
653 make_item(100.0, 50.0, "Monday"),
654 make_item(50.0, 100.0, "PDI"),
655 make_item(100.0, 102.0, "Personal Development"),
656 make_item(100.0, 111.0, "Intervention"),
657 make_item(100.0, 119.0, "10T7/Pd"),
658 make_item(100.0, 126.0, "MA7"),
659 make_item(100.0, 126.0, "Ms Test A"),
660 ];
661 let items: Vec<&TextItem> = src.iter().collect();
662
663 let lessons = parse_week_items(&items);
664 assert_eq!(lessons.len(), 1);
665 assert_eq!(lessons[0].subject, "PDI");
666 assert_eq!(lessons[0].class_code, "10T7/Pd");
667 assert_eq!(lessons[0].room, "MA7");
668 assert_eq!(lessons[0].teacher, "Ms Test A");
669 }
670
671 #[test]
672 fn parse_lesson_detects_two_digit_classcode() {
673 let src = [
674 make_item(100.0, 200.0, "Personal Development Intervention"),
675 make_item(100.0, 210.0, "10T7/Pd"),
676 make_item(100.0, 220.0, "Ms Test B"),
677 ];
678
679 let refs: Vec<&TextItem> = src.iter().collect();
680 let refsrefs: Vec<&&TextItem> = refs.iter().collect();
681
682 let lesson = parse_lesson_content(refsrefs, 1, 2);
683 assert_eq!(lesson.subject, "PDI");
684 assert_eq!(lesson.class_code, "10T7/Pd");
685 assert_eq!(lesson.teacher, "Ms Test B");
686 }
687
688 #[test]
689 fn extract_student_info_parens() {
690 let src = [make_item(10.0, 10.0, "Alex Testington (11XX)")];
691 let items: Vec<&TextItem> = src.iter().collect();
692
693 let (name, form) = extract_student_info(&items);
694 assert_eq!(name.unwrap(), "Alex Testington");
695 assert_eq!(form.unwrap(), "11XX");
696 }
697
698 #[test]
699 fn extract_student_info_separate() {
700 let src = [
701 make_item(10.0, 10.0, "Alex Testington"),
702 make_item(10.0, 12.0, "11XX"),
703 ];
704 let items: Vec<&TextItem> = src.iter().collect();
705
706 let (name, form) = extract_student_info(&items);
707 assert_eq!(name.unwrap(), "Alex Testington");
708 assert_eq!(form.unwrap(), "11XX");
709 }
710
711 #[test]
712 fn extract_student_info_parens_numeric() {
713 let src = [make_item(10.0, 10.0, "Alex Testington (917)")];
714 let items: Vec<&TextItem> = src.iter().collect();
715
716 let (name, form) = extract_student_info(&items);
717 assert_eq!(name.unwrap(), "Alex Testington");
718 assert_eq!(form.unwrap(), "917");
719 }
720
721 #[test]
722 fn extract_student_info_separate_alpha_num() {
723 let src = [
724 make_item(10.0, 10.0, "Ann Example"),
725 make_item(10.0, 13.0, "10A"),
726 ];
727 let items: Vec<&TextItem> = src.iter().collect();
728
729 let (name, form) = extract_student_info(&items);
730 assert_eq!(name.unwrap(), "Ann Example");
731 assert_eq!(form.unwrap(), "10A");
732 }
733}