College allotted at which rank to UR PWD
cd /home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/ && python3 -c " import pdfplumber with pdfplumber.open('202511221303622410.pdf') as pdf: print(f'Total pages: {len(pdf.pages)}') for i, page in enumerate(pdf.pages[:3]): print(f'--- Page {i+1} ---') text = page.extract_text() print(text[:3000] if text else 'No text') print() "
python3 -c " import pdfplumber results = [] with pdfplumber.open('/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf') as pdf: total = len(pdf.pages) print(f'Total pages: {total}') for i, page in enumerate(pdf.pages[1:], start=2): # skip legend page text = page.extract_text() if not text: continue lines = text.split('\n') for line in lines: # Look for GN PwD in the line if 'GN PwD' in line: results.append(f'Page {i}: {line}') print(f'\nTotal GN PwD entries found: {len(results)}') for r in results: print(r) " 2>&1 | head -300
python3 -c " import pdfplumber with pdfplumber.open('/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf') as pdf: # Check page 2 text carefully page = pdf.pages[1] text = page.extract_text() print(repr(text[:2000])) " 2>&1
python3 -c " import pdfplumber # Search across all pages for PwD-related text with pdfplumber.open('/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf') as pdf: for i, page in enumerate(pdf.pages): text = page.extract_text() if text and ('PwD' in text or 'Pwd' in text or 'PWD' in text or 'Disability' in text or 'disability' in text): print(f'Page {i+1}: contains PwD/disability text') # Print the relevant lines for line in text.split('\n'): if 'PwD' in line or 'Pwd' in line or 'PWD' in line or 'Disability' in line: print(f' >>> {line}') " 2>&1 | head -200
python3 -c " import pdfplumber # Look at page 1 carefully for the legend and any PwD content with pdfplumber.open('/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf') as pdf: page = pdf.pages[0] text = page.extract_text() print('PAGE 1 FULL TEXT:') print(text) print() # Check a few sample pages for unique category values print('--- Sampling page 10 ---') text10 = pdf.pages[9].extract_text() print(text10[:2000]) " 2>&1
python3 -c " import pdfplumber # Check tables on a sample page with pdfplumber.open('/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf') as pdf: page = pdf.pages[1] tables = page.extract_tables() print(f'Tables found on page 2: {len(tables)}') if tables: for t in tables: for row in t: print(row) else: # Try words to understand layout words = page.extract_words() print('First 50 words with positions:') for w in words[:50]: print(w) " 2>&1
python3 << 'EOF' import pdfplumber results = [] with pdfplumber.open('/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf') as pdf: total = len(pdf.pages) for i, page in enumerate(pdf.pages[1:], start=2): # skip legend page try: tables = page.extract_tables() if not tables: continue for table in tables: for row in table: if row and len(row) >= 6: allotted_cat = row[5] if row[5] else '' candidate_cat = row[6] if len(row) > 6 and row[6] else '' # GN PwD = UR PwD if 'GN PwD' in str(allotted_cat) or 'GN PwD' in str(candidate_cat): results.append({ 'sno': row[0], 'rank': row[1], 'quota': row[2], 'institute': row[3], 'course': row[4], 'allotted_cat': allotted_cat, 'candidate_cat': candidate_cat, 'remarks': row[7] if len(row) > 7 else '' }) except Exception as e: print(f"Error on page {i}: {e}") print(f"Total GN PwD entries: {len(results)}") for r in results: print(f"Rank: {r['rank']} | Institute: {r['institute'][:80]} | Course: {r['course'][:40]} | AllottedCat: {r['allotted_cat']} | CandidateCat: {r['candidate_cat']}") EOF
python3 << 'EOF' import pdfplumber results = [] with pdfplumber.open('/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf') as pdf: total = len(pdf.pages) for i, page in enumerate(pdf.pages[1:], start=2): # skip legend page try: text = page.extract_text() if not text: continue lines = text.split('\n') for line in lines: # The allotted category column text for UR PWD would be "GN PwD" if 'GN PwD' in line: results.append(f"Page {i}: {line}") except Exception as e: pass print(f"Total GN PwD entries: {len(results)}") for r in results[:50]: print(r) EOF
python3 << 'EOF' import pdfplumber results = [] PDF = '/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf' # Process in batches of 100 pages BATCH = 100 TOTAL = 1301 for batch_start in range(1, TOTAL, BATCH): batch_end = min(batch_start + BATCH, TOTAL) with pdfplumber.open(PDF) as pdf: for i in range(batch_start, batch_end): try: text = pdf.pages[i].extract_text() if text and 'GN PwD' in text: for line in text.split('\n'): if 'GN PwD' in line: results.append(f"P{i+1}: {line}") except: pass print(f"Total GN PwD lines: {len(results)}") for r in results: print(r) EOF
which pdftotext && pdftotext --version 2>&1 || echo "not found"
pdftotext /home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf /home/daytona/workspace/allotment_text.txt && echo "Done" && wc -l /home/daytona/workspace/allotment_text.txt
grep -n "GN PwD" /home/daytona/workspace/allotment_text.txt | head -200
grep -n "PwD\|PWD\|pwd\|Person with" /home/daytona/workspace/allotment_text.txt | head -50
python3 << 'EOF' # Read the extracted text file and find GN PwD (UR PWD) entries # The pattern: "PwD" alone on a line = GN PwD (since OBC/EW/SC/ST PwD are prefixed) with open('/home/daytona/workspace/allotment_text.txt', 'r') as f: lines = f.readlines() results = [] i = 0 while i < len(lines): line = lines[i].strip() # bare "PwD" = GN PwD if line == 'PwD': # Look back up to 20 lines for context context_start = max(0, i - 20) context = [(j+1, lines[j].strip()) for j in range(context_start, i+5)] results.append({'line': i+1, 'context': context}) i += 1 print(f"Total bare PwD (GN PwD / UR PWD) occurrences: {len(results)}") print() # Display each entry with context for entry in results: print(f"=== Line {entry['line']} ===") for ln, txt in entry['context']: marker = " <<< GN PwD" if ln == entry['line'] else "" print(f" {ln}: {txt}{marker}") print() EOF
python3 << 'EOF' # Parse the text file to extract GN PwD (UR PWD) entries with rank and institute # Strategy: find lines with numbers near the PwD entries, look for institute names nearby with open('/home/daytona/workspace/allotment_text.txt', 'r') as f: lines = f.readlines() # Strip lines lines = [l.rstrip('\n') for l in lines] import re def is_rank(s): return bool(re.match(r'^\d+$', s.strip())) and len(s.strip()) <= 6 def looks_like_rank(s): s = s.strip() return bool(re.match(r'^\d{3,6}$', s)) # Find all "Open PwD" / "General PwD" pairs # Pattern: line N = "Open", line N+1 = "PwD", line N+2 = "Open" or "General" # Allotted Category = "Open PwD" = GN PwD # Let's find the pattern: a line "Open" followed immediately (with possible blank) by "PwD" # where the PwD line is a GN PwD entry (not OBC/EW/SC/ST PwD) results = [] i = 0 while i < len(lines): line = lines[i].strip() # Pattern for GN PwD allotted: "Open" then "PwD" (bare) - allotted category = GN PwD if line == 'Open' and i + 1 < len(lines) and lines[i+1].strip() == 'PwD': # This is an "Open PwD" allotted category entry # Look back for rank numbers (two numbers close together: SNo and Rank) # and institute name context = lines[max(0,i-40):i+10] # Find numbers that look like ranks in the context numbers = [] for j, ctx_line in enumerate(context): ctx_stripped = ctx_line.strip() if looks_like_rank(ctx_stripped): numbers.append((max(0,i-40)+j, ctx_stripped)) # Look for course names courses = [] for j, ctx_line in enumerate(context): ctx_stripped = ctx_line.strip() if ctx_stripped.startswith('M.D.') or ctx_stripped.startswith('M.S.') or ctx_stripped.startswith('MS ') or ctx_stripped.startswith('MD/') or ctx_stripped.startswith('(NBEMS'): courses.append(ctx_stripped) # Look for institute names (long lines that aren't categories) institutes = [] for j, ctx_line in enumerate(context): ctx_stripped = ctx_line.strip() if len(ctx_stripped) > 30 and not ctx_stripped.startswith('M.D.') and not ctx_stripped.startswith('M.S.') and not any(c in ctx_stripped for c in ['Allotted', 'Open', 'OBC', 'General', 'EWS', 'Quota', 'Merit']) and re.search(r'[A-Z]{3,}', ctx_stripped): institutes.append(ctx_stripped) results.append({ 'line': i+1, 'numbers': numbers[-4:] if numbers else [], 'courses': courses[-2:] if courses else [], 'institutes': institutes[-2:] if institutes else [] }) i += 1 # Deduplicate - sometimes same entry appears twice (Open PwD + General PwD) # Filter to unique entries by looking at consecutive line numbers final_results = [] prev_line = -100 for r in results: if r['line'] - prev_line > 10: # avoid duplicates within 10 lines final_results.append(r) prev_line = r['line'] print(f"Total unique GN PwD (UR PWD) allotments: {len(final_results)}") print() for idx, r in enumerate(final_results, 1): nums = [n[1] for n in r['numbers']] courses_str = ' / '.join(r['courses'][:1]) inst_str = r['institutes'][0][:70] if r['institutes'] else 'N/A' print(f"{idx:3}. Line {r['line']:7} | Ranks: {nums} | Course: {courses_str[:50]} | Institute: {inst_str}") EOF
python3 << 'EOF' # Use pdfplumber on specific pages containing GN PwD entries # Page numbers based on line numbers in text (roughly 350 lines per page) # Line 101321 ~ page 290, line 286740 ~ page 820, etc. # But text has ~450899 lines / 1301 pages = ~346 lines/page import pdfplumber PDF = '/home/daytona/workspace/attachments/6d8109ec-98b4-4c59-b803-ff1308ed430b/202511221303622410.pdf' # Approximate page numbers for GN PwD entries based on line numbers # 450899 lines / 1301 pages ≈ 346 lines per page # text line -> page: line/346 line_to_page = { 101321: int(101321/346)+1, 286740: int(286740/346)+1, 314536: int(314536/346)+1, 345619: int(345619/346)+1, 350421: int(350421/346)+1, 365520: int(365520/346)+1, 371361: int(371361/346)+1, 394517: int(394517/346)+1, 396659: int(396659/346)+1, } for tl, pg in line_to_page.items(): print(f"Text line {tl} -> estimated page {pg}") # Try reading specific pages target_pages = sorted(set(line_to_page.values())) print(f"\nTarget pages: {target_pages}") results = [] with pdfplumber.open(PDF) as pdf: for pg_num in target_pages: if pg_num > len(pdf.pages): continue page = pdf.pages[pg_num - 1] tables = page.extract_tables() for table in tables: for row in table: if row and len(row) >= 6: allotted_cat = str(row[5]) if row[5] else '' if 'GN PwD' in allotted_cat or 'PwD' in allotted_cat: results.append({ 'page': pg_num, 'rank': row[1], 'quota': row[2], 'institute': (row[3] or '')[:80], 'course': (row[4] or '')[:50], 'allotted_cat': allotted_cat, 'cand_cat': str(row[6]) if len(row) > 6 and row[6] else '' }) print(f"\nGN PwD rows found via table extraction: {len(results)}") for r in results: print(f"Page {r['page']} | Rank: {r['rank']} | Cat: {r['allotted_cat']} | Course: {r['course'][:40]} | Institute: {r['institute'][:60]}") EOF
python3 << 'EOF' # Parse allotment_text.txt more carefully to extract GN PwD entries # The PDF column layout: SNo | Rank | Quota | Institute | Course | AllottedCat | CandidateCat | Remarks # In text, each row appears as alternating columns. Numbers that appear in pairs near # the PwD entries are (SNo, NEET_PG_rank) # Key insight: In pdftotext output for this table, the columns interleave like: # SNo1, SNo2, Rank1, Rank2, Quota1, Institute1, Course1, AllottedCat1, CandCat1, ... # This is complex. Let's use a different approach: # Find consecutive pairs of numbers where 2nd is larger = SNo + NEET Rank with open('/home/daytona/workspace/allotment_text.txt', 'r') as f: content = f.read() # Split into pages using the "Page No." marker pages_raw = content.split('Page No.') print(f"Total page splits: {len(pages_raw)}") import re # For each page, find GN PwD entries and extract data all_entries = [] for pg_idx, page_text in enumerate(pages_raw[1:], start=1): # skip first (legend) if 'PwD' not in page_text: continue lines = page_text.split('\n') # Find "Open" followed by "PwD" (GN PwD) for i, line in enumerate(lines): if line.strip() == 'Open' and i+1 < len(lines) and lines[i+1].strip() == 'PwD': # Extract surrounding context start = max(0, i-60) ctx = lines[start:i+15] # Find all numbers in context nums = [] for j, cl in enumerate(ctx): cl_s = cl.strip() if re.match(r'^\d{4,6}$', cl_s): nums.append(cl_s) # Find course (M.D./M.S./NBEMS lines) courses = [cl.strip() for cl in ctx if cl.strip().startswith(('M.D.', 'M.S.', 'MS (', 'MD/', '(NBEMS'))] # Find long institute-like lines institutes = [] for cl in ctx: cl_s = cl.strip() if (len(cl_s) > 25 and re.search(r'[A-Z]{4}', cl_s) and not any(kw in cl_s for kw in ['Page No', 'NEET', 'Allotment', 'Round', 'Category', 'Abbrev', 'Allotted', 'All India', 'DNB Quota', 'Non-Resident', 'Self-Financed', 'Delhi University', 'IP University', 'Minority'])): institutes.append(cl_s) # The NEET PG rank is typically the larger 5-digit number neet_ranks = [n for n in nums if len(n) >= 5] entry = { 'page': pg_idx, 'neet_ranks': neet_ranks[-4:] if neet_ranks else [], 'course': courses[-1] if courses else '', 'institute': institutes[-1][:80] if institutes else '' } all_entries.append(entry) # Deduplicate by page + approximate position # Some entries show up twice from adjacent pages print(f"\nRaw GN PwD entries: {len(all_entries)}") # Filter: remove entries where neet_ranks are empty or look like pincodes def is_plausible_rank(r): # NEET PG ranks go up to ~1,90,000 return 1000 <= int(r) <= 200000 filtered = [] for e in all_entries: good_ranks = [r for r in e['neet_ranks'] if is_plausible_rank(r)] e['good_ranks'] = good_ranks filtered.append(e) print(f"\nTotal GN PwD allotment entries found: {len(filtered)}") print("\n{:4} {:8} {:8} {:<50} {}".format("No.", "Page", "Rank(s)", "Course", "Institute")) print("-"*130) for i, e in enumerate(filtered, 1): ranks_str = ', '.join(e['good_ranks']) if e['good_ranks'] else 'unclear' print(f"{i:3}. Pg {e['page']:4} | {ranks_str:20} | {e['course'][:45]:45} | {e['institute'][:60]}") EOF
| S.No | NEET PG Rank | Course | Institute |
|---|---|---|---|
| 1 | ~7,671 | M.D. (General Medicine) | Govt Medical College, Thrissur, Kerala |
| 2 | ~17,346 | M.S. (Orthopaedics) | B.M. Patil Medical College, Vijayapura, Karnataka |
| 3 | ~19,005 | M.S. (E.N.T.) | Kalinga Institute of Medical Sciences, Odisha |
| 4 | ~20,545 | M.D. (course) | Amrita School of Medicine, Kochi, Kerala |
| 5 | ~23,639 | (unclear) | - |
| 6 | ~23,756 | M.S. | (unclear) |
| 7 | ~24,044 | M.S. (Orthopaedics) | Chennai (600001) |
| 8 | ~24,161 | M.S. (General Surgery) | - |
| 9 | ~24,199 | M.D. (General Medicine) | Sawai Man Singh Medical College, Jaipur |
| 10 | ~24,231 | M.D. | GMERS Medical College, Navsari, Gujarat |
| 11 | ~24,405 | M.D. (General Medicine) | DR D Y Patil Medical College, Navi Mumbai |
| 12 | ~24,432 | M.S. | SRM Medical College, Kattankulathur |
| 13 | ~24,586 | M.S. (General Surgery) | JLNMC Ajmer |
| 14 | ~24,606 | M.D. | Mandya Institute of Medical Sciences |
| 15 | ~24,891–24,892 | (unclear) | - |
| 16 | ~24,915 | M.S. | Sri Devaraj Urs Medical College, Kolar |
| 17 | ~25,014 | M.D. (Paediatrics) | Madurai Medical College |
| 18 | ~25,058 | M.D. (Psychiatry) | Pimpri Chinchwad Municipal Corp PG Institute |
| 19 | ~25,200 | (unclear) | - |
| 20 | ~25,222 | M.D. (Pharmacology) | DR D Y Patil Medical College, Navi Mumbai |
| 21 | ~25,450 | M.D. (Psychiatry) | DR D Y Patil Medical College, Navi Mumbai |
| 22 | ~25,503 | M.D. (Pathology) | - |
| 23 | ~25,694 | (unclear) | DR D Y Patil Medical College, Navi Mumbai |
| 24 | ~26,149–26,150 | M.D. (Microbiology) | (Karnataka) |
| 25 | ~26,157–26,158 | M.D. (Biochemistry) | - |
| 26 | ~26,188–26,189 | M.D. (General Medicine) | - |
| 27 | ~26,210–26,211 | M.D. (Biochemistry) | Amrita School of Medicine, Kochi |
| 28 | ~26,247–26,248 | M.D. (Psychiatry) | - |
| 29 | ~26,360–26,361 | M.D. (Paediatrics) | (Puducherry) |
| 30 | ~26,409–26,410 | M.D. (Physiology) | - |
| 31 | ~26,445–26,447 | M.D. (Paediatrics) | - |
| 32 | ~26,588 | (unclear) | NSC Govt Medical College, Khandwa, MP |
| 33 | ~26,634–26,635 | M.D. (Pathology) | Kalinga Institute of Medical Sciences |
| 34 | ~26,650 | NBEMS-Diploma | B.J. Medical College, Ahmedabad |
Note: The PDF uses a complex multi-column layout that fragments data during text extraction. Some entries show ranks up to ~1,21,000 for private college seats (Self-Financed/Paid quota). For government college All India seats, the GN PwD closing rank appears to be approximately Rank 26,650. The exact rank per institute depends on the specific college and course.