medgemma works
This commit is contained in:
parent
6f052be127
commit
4ee0ad87bb
10 changed files with 1270 additions and 0 deletions
2
.gitignore
vendored
2
.gitignore
vendored
|
|
@ -167,6 +167,8 @@ components
|
||||||
.vscode/
|
.vscode/
|
||||||
whoosh_index/
|
whoosh_index/
|
||||||
|
|
||||||
|
*.csv
|
||||||
|
*.log
|
||||||
*.pdf
|
*.pdf
|
||||||
*.seg
|
*.seg
|
||||||
*.sql
|
*.sql
|
||||||
|
|
|
||||||
16
2026/export-all-patho.py
Normal file
16
2026/export-all-patho.py
Normal file
|
|
@ -0,0 +1,16 @@
|
||||||
|
import pandas as pd
|
||||||
|
import pymysql
|
||||||
|
import os
|
||||||
|
|
||||||
|
# Configure database connection
|
||||||
|
conn = pymysql.connect(host='localhost', user='ntuh', password='n122119493', db='adm15')
|
||||||
|
|
||||||
|
# Execute query and load into a pandas DataFrame
|
||||||
|
df = pd.read_sql_query("SELECT * FROM registry_pathologyreport", conn)
|
||||||
|
|
||||||
|
# Export DataFrame to an Excel file
|
||||||
|
# output_file = '/data/patho-2026.xlsx'
|
||||||
|
# df.to_excel(output_file, index=False)
|
||||||
|
output_file_parquet = '/data/patho-2026.parquet'
|
||||||
|
df.to_parquet(output_file_parquet, index=False, engine='pyarrow')
|
||||||
|
print(f"Data exported to {output_file_parquet}")
|
||||||
Binary file not shown.
|
Before Width: | Height: | Size: 89 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 33 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 126 KiB |
435
patho/0727/extract.py
Executable file
435
patho/0727/extract.py
Executable file
|
|
@ -0,0 +1,435 @@
|
||||||
|
#!/usr/bin/env python
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import csv
|
||||||
|
import re
|
||||||
|
import openpyxl
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
base_dir = os.path.dirname(script_dir)
|
||||||
|
|
||||||
|
|
||||||
|
def is_valid_tissue(t):
|
||||||
|
if not t or not isinstance(t, str):
|
||||||
|
return False
|
||||||
|
t = t.strip()
|
||||||
|
clean_t = re.sub(r'[\.,\-\_\s\(\)\"\']', '', t)
|
||||||
|
clean_no_dim = re.sub(r'\d+[\.\*x\d]*cm', '', clean_t, flags=re.IGNORECASE)
|
||||||
|
if not clean_t or len(clean_t) <= 2 or clean_t.lower() in ['none', 'nil']:
|
||||||
|
return False
|
||||||
|
if clean_t.upper() in ['BM', 'BX', 'CX', 'HX', 'PX', 'DX', 'NA', 'NB']:
|
||||||
|
return False
|
||||||
|
tumor_terms = ['thrombus', 'clot', 'hematoma', 'abscess', 'carcinoma', 'adenocarcinoma', 'adenoma', 'polyp', 'cyst', 'synovialcyst', 'tumor', 'tumour', 'mass', 'nodule', 'glioma', 'astrocytoma', 'meningioma', 'schwannoma', 'lymphoma', 'melanoma']
|
||||||
|
if clean_t.lower() in tumor_terms or any(clean_t.lower().endswith(k) for k in ['cyst', 'tumor', 'tumour', 'glioma']) or any(clean_no_dim.lower().endswith(k) for k in ['cyst', 'tumor', 'tumour', 'glioma']) or 'subcutaneoustumor' in clean_t.lower() or 'intraduraltumor' in clean_t.lower() or 'hyperintensity' in clean_t.lower() or 'hypointensity' in clean_t.lower() or 'duraseeding' in clean_t.lower() or 'cavitytumor' in clean_t.lower():
|
||||||
|
return False
|
||||||
|
if '位置不明' in t or 'sitenotstate' in clean_t.lower() or 'siteunspecified' in clean_t.lower():
|
||||||
|
return False
|
||||||
|
if any(k in clean_t.lower() for k in ['recurrenttumor', 'granulationtis', 'suspecttumor', 'rotumor', 'ronecros']):
|
||||||
|
return False
|
||||||
|
invalid_keywords = ['送檢單位', '切取日期', '固定日期', '檢查項目', '病理報告', '病歷號', '版本:', 'Result:', 'Comments:']
|
||||||
|
if any(kw in t for kw in invalid_keywords):
|
||||||
|
return False
|
||||||
|
if any(t.startswith(prefix) for prefix in ['送檢', '切取', '檢查', '病理', '病歷', '版本']):
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def extract_tissue(html_content, path_code=''):
|
||||||
|
if not html_content or not isinstance(html_content, str):
|
||||||
|
return ""
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html_content, 'html.parser')
|
||||||
|
|
||||||
|
tissue = ""
|
||||||
|
# 1. Check lblTissue span
|
||||||
|
span = soup.find('span', id=lambda x: x and 'lblTissue' in x)
|
||||||
|
if span and is_valid_tissue(span.text):
|
||||||
|
tissue = span.text.strip()
|
||||||
|
|
||||||
|
# 2. Check table row with 組織由來
|
||||||
|
if not tissue:
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
ths = [th.text.strip() for th in tr.find_all('th', recursive=False)]
|
||||||
|
tds = [td.text.strip() for td in tr.find_all('td', recursive=False)]
|
||||||
|
if '組織由來' in ths:
|
||||||
|
idx = ths.index('組織由來')
|
||||||
|
if idx < len(tds) and is_valid_tissue(tds[idx]):
|
||||||
|
tissue = tds[idx]
|
||||||
|
break
|
||||||
|
elif '組織由來' in tr.text:
|
||||||
|
if tds and is_valid_tissue(tds[-1]):
|
||||||
|
tissue = tds[-1]
|
||||||
|
|
||||||
|
# 3. Check 器官與術式 / 器官 / 部位 / 檢體部位
|
||||||
|
if not tissue:
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
ths = [th.text.strip() for th in tr.find_all('th', recursive=False)]
|
||||||
|
tds = [td.text.strip() for td in tr.find_all('td', recursive=False)]
|
||||||
|
for header in ['器官與術式', '器官', '部位', '檢體部位']:
|
||||||
|
if header in ths:
|
||||||
|
idx = ths.index(header)
|
||||||
|
if idx < len(tds) and is_valid_tissue(tds[idx]):
|
||||||
|
tissue = tds[idx]
|
||||||
|
break
|
||||||
|
if tissue:
|
||||||
|
break
|
||||||
|
|
||||||
|
# 4. Dig site from 組織學診斷 row if tissue is still empty
|
||||||
|
if not tissue:
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
ths = [th.text.strip() for th in tr.find_all('th', recursive=False)]
|
||||||
|
tds = [td.text.strip() for td in tr.find_all('td', recursive=False)]
|
||||||
|
if any('組織學診斷' in th for th in ths):
|
||||||
|
if tds:
|
||||||
|
val = tds[-1]
|
||||||
|
if val:
|
||||||
|
s = re.sub(r'^\d+[\.\s]*', '', val)
|
||||||
|
split_site = re.split(r',\s*(?:\d+(?:\-\d+)?\s*cm(?:\s+AAV|\s+above\s+anal\s+verge|\s+[A-Za-z]+)?|AAV|above\s+anal\s+verge|tumor\s+resection|tumor\s+excision|metastasis\s+excision|effusion\s+drainage|drainage|reconstruction|debridement|spondylectomy|laminectomy|discectomy|arthroplasty|fixation|fusion|stripping|incisional(?:\s+biopsy)?|excisional(?:\s+biopsy)?|punch(?:\s+biopsy)?|shave(?:\s+biopsy)?|wedge(?:\s+biopsy)?|core(?:\s+biopsy)?|site\s+not\s+state(?:d)?|site\s+unspecified|\?\s*side|\?|craniotomy(?:\s+and\s+|\s*\&\s*|\s+)excision|craniotomy|FESS|transphenoid|transsphenoidal|panendoscopic|choledochoscopic|colonoscopic|endoscopical|endoscopic|laparoscopic|transcervical|open|biopsy|polypectomy|(?:partial|subtotal|radical|wide|total)?\s*resection|excision|curettage|aspiration|dissection|cholecystectomy|hysterectomy|nephrectomy|lobectomy|mastectomy|meningioma|lipoma|fibroma|adenoma|carcinoma|polyp|hyperplasia|cyst|gastritis|sinusitis|cholecystectomy|fibrosis)\b', s, flags=re.IGNORECASE)[0]
|
||||||
|
split_site = re.split(r',\s*(?:corpus|cervix)\b', split_site, flags=re.IGNORECASE)[0]
|
||||||
|
if is_valid_tissue(split_site) and split_site != val:
|
||||||
|
tissue = split_site.strip()
|
||||||
|
break
|
||||||
|
|
||||||
|
# 5. Check active report tab link: <a href="...contentHolder" title="...">
|
||||||
|
if not tissue:
|
||||||
|
a_tag = soup.find('a', href=lambda x: x and 'contentHolder' in x, title=True)
|
||||||
|
if a_tag and is_valid_tissue(a_tag.get('title')):
|
||||||
|
tissue = a_tag['title'].strip()
|
||||||
|
|
||||||
|
# 6. Check selected option: <option selected="selected">... (tissue)</option>
|
||||||
|
if not tissue:
|
||||||
|
opt = soup.find('option', selected=True)
|
||||||
|
if opt and opt.text:
|
||||||
|
m = re.search(r'\(([^()]+)\)$', opt.text)
|
||||||
|
if m and is_valid_tissue(m.group(1)):
|
||||||
|
tissue = m.group(1).strip()
|
||||||
|
|
||||||
|
# 7. Dig from 檢查報告 (lblResult or textarea or full text) if still blank/dot
|
||||||
|
if not tissue:
|
||||||
|
ta = soup.find('textarea')
|
||||||
|
res_span = soup.find('span', id=lambda x: x and 'lblResult' in x)
|
||||||
|
raw_text = (ta.get_text(separator='\n') if ta else (res_span.get_text(separator='\n') if res_span else soup.get_text(separator='\n'))).strip()
|
||||||
|
if raw_text:
|
||||||
|
m = re.search(r'(?:Pathology Number of specimen|Pathology Number|Specimen Number|sample number|specimen number|from sample|from specimen|sample|specimen)\s*[::]\s*([A-Za-z0-9_\-]+)|(?:from sample|from specimen|sample number|specimen number)\s*[::\s]\s*([A-Za-z0-9_\-]+)', raw_text, re.IGNORECASE)
|
||||||
|
matched_code = (m.group(1) or m.group(2)).strip() if m else None
|
||||||
|
if matched_code:
|
||||||
|
tissue = matched_code
|
||||||
|
else:
|
||||||
|
lines = [l.strip() for l in raw_text.splitlines() if l.strip()]
|
||||||
|
if lines:
|
||||||
|
first_line = lines[0]
|
||||||
|
if not any(k in first_line for k in ['MACROSCOPIC', 'MICROSCOPIC', 'The specimen', 'Grossly']):
|
||||||
|
split_site = re.split(r',\s*(?:\d+(?:\-\d+)?\s*cm(?:\s+AAV|\s+above\s+anal\s+verge|\s+[A-Za-z]+)?|AAV|above\s+anal\s+verge)\b', first_line, flags=re.IGNORECASE)[0]
|
||||||
|
split_site = re.split(r',\s*(?:[A-Za-z\s]+)?(?:tumor\s+resection|tumor\s+excision|spondylectomy|laminectomy|discectomy|thrombus|clot|hematoma|abscess|echo-guided|ct-guided|us-guided|ultrasound-guided|incisional|excisional|punch|shave|wedge|core|needle|fine\s+needle|colonoscopic|endoscopic|laparoscopic|transcervical|open)?\s*(?:biopsy|aspiration|fna|polypectomy|resection|excision|curettage|dissection|reconstruction|debridement|drainage|tumor\s+resection|tumor\s+excision|spondylectomy|laminectomy|discectomy|tubular adenoma|adenocarcinoma|carcinoma|polyp|abscess|thrombus|clot|hematoma|necrosis)\b[^\n\r,]*', first_line, flags=re.IGNORECASE)[0]
|
||||||
|
split_site = re.split(r',\s*labeled\s+as\b', split_site, flags=re.IGNORECASE)[0]
|
||||||
|
if is_valid_tissue(split_site):
|
||||||
|
tissue = split_site.strip()
|
||||||
|
|
||||||
|
# If tissue is LN or a reference code like S2555571, expand or check description
|
||||||
|
if tissue.upper() == 'LN':
|
||||||
|
tissue = 'Lymph node'
|
||||||
|
|
||||||
|
if not tissue:
|
||||||
|
span = soup.find('span', id=lambda x: x and 'lblTissue' in x)
|
||||||
|
if span and span.text.strip().upper() == 'LN':
|
||||||
|
tissue = 'Lymph node'
|
||||||
|
|
||||||
|
if tissue and re.match(r'^[A-Za-z0-9_\-]+$', tissue):
|
||||||
|
ta = soup.find('textarea')
|
||||||
|
res_span = soup.find('span', id=lambda x: x and 'lblResult' in x)
|
||||||
|
raw_text = (ta.get_text(separator='\n') if ta else (res_span.get_text(separator='\n') if res_span else soup.get_text(separator='\n'))).strip()
|
||||||
|
if raw_text:
|
||||||
|
m = re.search(rf'{re.escape(tissue)}\s+([A-Za-z][^\n\r]+)', raw_text)
|
||||||
|
if m:
|
||||||
|
desc = m.group(1).strip()
|
||||||
|
split_site = re.split(r',\s*(?:tumor\s+resection|tumor\s+excision|excision|resection|biopsy|carcinoma|adenoma|metastatic)\b', desc, flags=re.IGNORECASE)[0]
|
||||||
|
if is_valid_tissue(split_site) and split_site != desc and not re.match(r'^[A-Za-z0-9_\-]+$', split_site):
|
||||||
|
tissue = split_site.strip()
|
||||||
|
|
||||||
|
return tissue
|
||||||
|
|
||||||
|
|
||||||
|
def extract_short_diagnosis(html_content, path_code=''):
|
||||||
|
if not html_content or not isinstance(html_content, str):
|
||||||
|
return ""
|
||||||
|
|
||||||
|
if str(path_code).strip() in ['S2605800', 'S2602629'] or any(code in str(html_content) for code in ['S2605800', 'S2602629']):
|
||||||
|
soup = BeautifulSoup(html_content, 'html.parser')
|
||||||
|
ta = soup.find('textarea')
|
||||||
|
if ta:
|
||||||
|
lines = [l.strip() for l in ta.text.splitlines() if l.strip()]
|
||||||
|
if lines:
|
||||||
|
return lines[0]
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html_content, 'html.parser')
|
||||||
|
|
||||||
|
# 1. Check table rows for 組織學診斷 / 診斷描述
|
||||||
|
table_diags = []
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
ths = [th.text.strip() for th in tr.find_all('th', recursive=False)]
|
||||||
|
tds = [td.text.strip() for td in tr.find_all('td', recursive=False)]
|
||||||
|
for h in ths:
|
||||||
|
if any(k in h for k in ['組織學診斷', '診斷描述']):
|
||||||
|
idx = ths.index(h)
|
||||||
|
if idx < len(tds):
|
||||||
|
val = tds[idx].text.strip() if hasattr(tds[idx], 'text') else str(tds[idx]).strip()
|
||||||
|
if val and val not in table_diags:
|
||||||
|
table_diags.append(val)
|
||||||
|
|
||||||
|
if not table_diags:
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
cells = [td.text.strip() for td in tr.find_all(['th', 'td'])]
|
||||||
|
text = ' '.join(cells)
|
||||||
|
if any(k in text for k in ['組織學診斷', '診斷描述']):
|
||||||
|
tds = tr.find_all('td')
|
||||||
|
if tds:
|
||||||
|
val = tds[-1].text.strip()
|
||||||
|
if val and val not in table_diags:
|
||||||
|
table_diags.append(val)
|
||||||
|
|
||||||
|
if table_diags:
|
||||||
|
return ', '.join(table_diags)
|
||||||
|
|
||||||
|
# 2. Check textarea or lblResult
|
||||||
|
ta = soup.find('textarea')
|
||||||
|
res_span = soup.find('span', id=lambda x: x and 'lblResult' in x)
|
||||||
|
raw_text = (ta.text if ta else (res_span.text if res_span else '')).strip()
|
||||||
|
|
||||||
|
if raw_text:
|
||||||
|
# Check explicit Histological diagnosis field in text
|
||||||
|
m_histo = re.search(r'Histological diagnosis[::\s]\s*([^\*\n\r]+)', raw_text, re.IGNORECASE)
|
||||||
|
if m_histo and m_histo.group(1).strip():
|
||||||
|
return m_histo.group(1).strip()
|
||||||
|
|
||||||
|
# Special handling for Molecular / Special laboratory reports starting with 送檢單位 or 檢查項目
|
||||||
|
if '送檢單位' in raw_text or '檢查項目' in raw_text:
|
||||||
|
if '檢查項目:' in raw_text or '檢查項目:' in raw_text:
|
||||||
|
m = re.search(r'檢查項目[::]\s*(.*?)(?:Pathology Number|Probe:|Control:|Duration|Result:|檢驗報告|送檢|\n|\r|\s{4,}|$)', raw_text, re.DOTALL)
|
||||||
|
if m and m.group(1).strip():
|
||||||
|
item = m.group(1).strip()
|
||||||
|
m_res = re.search(r'(?:Result:|檢驗報告[::])\s*(.*?)(?:Procedures:|Comments:|參考文獻|Ref:|\n\n|\r\n\r\n|\s{4,}|$)', raw_text, re.DOTALL)
|
||||||
|
if m_res and m_res.group(1).strip():
|
||||||
|
res_val = m_res.group(1).strip().replace('\n', ' ').replace('\t', ' ')
|
||||||
|
res_val = re.sub(r'\s{2,}', ' ', res_val)
|
||||||
|
return f'{item}; Result: {res_val}'
|
||||||
|
return item
|
||||||
|
elif '檢驗報告' in raw_text or 'Result:' in raw_text:
|
||||||
|
m_res = re.search(r'(?:Result:|檢驗報告[::])\s*(.*?)(?:Procedures:|Comments:|參考文獻|Ref:|\n\n|\r\n\r\n|\s{4,}|$)', raw_text, re.DOTALL)
|
||||||
|
if m_res and m_res.group(1).strip():
|
||||||
|
res_val = m_res.group(1).strip().replace('\n', ' ').replace('\t', ' ')
|
||||||
|
return re.sub(r'\s{2,}', ' ', res_val)
|
||||||
|
|
||||||
|
# Standard Pathology Report
|
||||||
|
cut_patterns = [
|
||||||
|
r'The specimen submitted', r'Grossly', r'Microscopically', r'Pathology Number',
|
||||||
|
r'送檢單位', r'檢查項目', r'Result:', r'Comments:', r'Ref:', r'\[\d+\]\.'
|
||||||
|
]
|
||||||
|
text_cut = re.split('|'.join(cut_patterns), raw_text, flags=re.IGNORECASE)[0]
|
||||||
|
parts = re.split(r'\n|\r|\t|\s{2,}', text_cut)
|
||||||
|
non_empty = [p.strip() for p in parts if p.strip()]
|
||||||
|
if non_empty:
|
||||||
|
first_p = non_empty[0]
|
||||||
|
m_end = re.search(r',\s*([A-Z][A-Za-z\s\-]+(?:adenoma|carcinoma|polyp|hyperplasia|gastritis|inflammation|fibrosis|cyst|lipoma|lymphoma))$', first_p, re.IGNORECASE)
|
||||||
|
if m_end:
|
||||||
|
return m_end.group(1).strip()
|
||||||
|
return first_p
|
||||||
|
|
||||||
|
# 3. Fallback to lblBedDiagnosis if available
|
||||||
|
bed_span = soup.find('span', id=lambda x: x and 'lblBedDiagnosis' in x)
|
||||||
|
if bed_span and bed_span.text.strip():
|
||||||
|
return bed_span.text.strip()
|
||||||
|
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def classify_site(tissue, diagnosis, html=''):
|
||||||
|
site_text = (str(tissue) + ' ' + str(diagnosis)).lower()
|
||||||
|
site_text = re.sub(r'\bpd\-?l1\b', '', site_text, flags=re.IGNORECASE)
|
||||||
|
|
||||||
|
spinal_kw = [
|
||||||
|
'spine', 'spinal', 'vertebra', 'vertebrae', 'vertebral', 'lumbar', 'thoracic', 'cervical',
|
||||||
|
'sacral', 'sacrum', 'epidural', 'paraspinal', 'disc', 'spondylectomy', 'laminectomy', 'discectomy',
|
||||||
|
'ligamentum flavum', 'ligamentum', 'ulbd', 'neural placode', 'placode'
|
||||||
|
]
|
||||||
|
spinal_patterns = [r'\b' + k + r'\b' for k in spinal_kw] + [r'\b[ltcs]\d+(?:[\-\/]\d+)?\b']
|
||||||
|
|
||||||
|
cranial_kw = [
|
||||||
|
'brain', 'brainstem', 'cerebrum', 'cerebral', 'cerebellar', 'cerebellum', 'pituitary', 'cranial', 'cranium',
|
||||||
|
'skull', 'dura', 'dural', 'temporal', 'frontal', 'parietal', 'occipital', 'ventricle',
|
||||||
|
'paranasal', 'head', 'scalp', 'orbit', 'sellar', 'suprasellar', 'cerebellopontine', 'cp angle', 'cpangle',
|
||||||
|
'meninx', 'meninges', 'sphenoid', 'sinonasal', 'nasal cavity', 'meningioma', 'forehead'
|
||||||
|
]
|
||||||
|
cranial_patterns = [r'\b' + k + r'\b' for k in cranial_kw] + [r'\bcp\s+angle\b', r'\bnasal\s+cavity\b']
|
||||||
|
|
||||||
|
is_spinal = any(re.search(p, site_text) for p in spinal_patterns)
|
||||||
|
is_cranial = any(re.search(p, site_text) for p in cranial_patterns)
|
||||||
|
|
||||||
|
if 'dura mater' in site_text or 'dural' in site_text and 'epidural' not in site_text:
|
||||||
|
is_cranial = True
|
||||||
|
|
||||||
|
if is_spinal and not is_cranial:
|
||||||
|
return 'spinal'
|
||||||
|
elif is_cranial and not is_spinal:
|
||||||
|
return 'cranial'
|
||||||
|
elif is_spinal and is_cranial:
|
||||||
|
if any(k in site_text for k in ['brain', 'cerebrum', 'dura mater']):
|
||||||
|
return 'cranial'
|
||||||
|
return 'spinal'
|
||||||
|
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
|
||||||
|
def classify_type(tissue, diagnosis, html=''):
|
||||||
|
diag_str = str(diagnosis).lower()
|
||||||
|
text = (str(tissue) + ' ' + diag_str).lower()
|
||||||
|
|
||||||
|
if any(k in diag_str for k in ['no evidence of', 'negative for', 'free of', 'no tumor', 'without tumor', 'no residual']):
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
if any(k in diag_str for k in ['necrotic tissue', 'necrosis', 'organizing hematoma', 'hematoma', 'thrombus', 'clot', 'abscess', 'infarction', 'fibrosis', 'sinusitis', 'cholecystitis', 'gastritis', 'varicose']):
|
||||||
|
definitive_tumor = ['carcinoma', 'adenocarcinoma', 'adenoma', 'sarcoma', 'osteosarcoma', 'osteoma', 'astrocytoma', 'glioma', 'meningioma', 'glioblastoma', 'subependymoma', 'ependymoma', 'lipoma', 'fibroma', 'papilloma', 'lymphoma', 'melanoma', 'schwannoma', 'malignancy', 'malignant']
|
||||||
|
if not any(k in diag_str for k in definitive_tumor):
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
non_tumor_findings = ['granuloma', 'inflammatory cell infiltration', 'organizing hematoma', 'hematoma', 'thrombus', 'clot', 'abscess', 'necrosis', 'infarction', 'fibrosis', 'sinusitis', 'cholecystitis', 'gastritis', 'varicose', 'amyloid angiopathy']
|
||||||
|
neoplastic_findings = [
|
||||||
|
'carcinoma', 'adenocarcinoma', 'adenoma', 'sarcoma', 'osteosarcoma', 'osteoma', 'chondroma', 'hemangioma', 'neuroma', 'leiomyoma', 'myxoma',
|
||||||
|
'astrocytoma', 'glioma', 'meningioma', 'glioblastoma', 'subependymoma', 'ependymoma', 'ganglioglioma', 'oligodendroglioma', 'medulloblastoma', 'chordoma', 'hemangioblastoma', 'craniopharyngioma', 'lipoma', 'fibroma', 'papilloma',
|
||||||
|
'polyp', 'hyperplasia', 'metastasis', 'metastatic', 'malignancy', 'malignant',
|
||||||
|
'schwannoma', 'neurofibroma', 'melanoma', 'lymphoma', 'plasmacytosis', 'neoplastic', 'who grade', 'cns who grade'
|
||||||
|
]
|
||||||
|
|
||||||
|
if any(k in diag_str for k in non_tumor_findings) and not any(k in diag_str for k in neoplastic_findings):
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
tumor_kw = [
|
||||||
|
'malignancy', 'malignant', 'osteoma', 'chondroma', 'hemangioma', 'neuroma', 'leiomyoma', 'myxoma',
|
||||||
|
'subependymoma', 'ependymoma', 'ganglioglioma', 'oligodendroglioma', 'medulloblastoma', 'chordoma', 'hemangioblastoma', 'craniopharyngioma', 'who grade', 'cns who grade', 'tumor', 'tumour', 'carcinoma', 'adenocarcinoma', 'adenoma', 'sarcoma', 'osteosarcoma',
|
||||||
|
'astrocytoma', 'glioma', 'meningioma', 'glioblastoma', 'lipoma', 'fibroma', 'papilloma', 'cyst',
|
||||||
|
'polyp', 'hyperplasia', 'mass', 'lesion', 'neoplasm', 'metastasis', 'metastatic',
|
||||||
|
'schwannoma', 'neurofibroma', 'melanoma', 'lymphoma', 'plasmacytosis', 'neoplastic'
|
||||||
|
]
|
||||||
|
|
||||||
|
if any(k in text for k in tumor_kw):
|
||||||
|
return 'tumor'
|
||||||
|
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
|
||||||
|
def process_extraction(input_excel=None, output_csv=None, limit=None, mode='all'):
|
||||||
|
if input_excel is None:
|
||||||
|
input_excel = os.path.join(script_dir, 'registry_pathologyreport.xlsx')
|
||||||
|
if not os.path.exists(input_excel) and os.path.exists('registry_pathologyreport.xlsx'):
|
||||||
|
input_excel = 'registry_pathologyreport.xlsx'
|
||||||
|
|
||||||
|
if output_csv is None:
|
||||||
|
output_csv = os.path.join(script_dir, 'registry_pathologyreport_extracted.csv')
|
||||||
|
|
||||||
|
print(f"Opening Excel file: {input_excel} ...")
|
||||||
|
wb = openpyxl.load_workbook(input_excel, read_only=True)
|
||||||
|
ws = wb.active
|
||||||
|
|
||||||
|
headers = None
|
||||||
|
chart_idx = None
|
||||||
|
path_idx = None
|
||||||
|
html_idx = None
|
||||||
|
|
||||||
|
all_raw_rows = []
|
||||||
|
|
||||||
|
print(f"Reading Excel rows...")
|
||||||
|
for i, row in enumerate(ws.iter_rows()):
|
||||||
|
vals = [cell.value for cell in row]
|
||||||
|
if headers is None:
|
||||||
|
headers = vals
|
||||||
|
chart_idx = headers.index('ChartNo')
|
||||||
|
path_idx = headers.index('PathCode')
|
||||||
|
html_idx = headers.index('html')
|
||||||
|
continue
|
||||||
|
|
||||||
|
all_raw_rows.append(vals)
|
||||||
|
|
||||||
|
if mode == 'last' and limit:
|
||||||
|
raw_rows = all_raw_rows[-limit:]
|
||||||
|
print(f"Processing last {len(raw_rows)} rows (out of {len(all_raw_rows)} total)...")
|
||||||
|
else:
|
||||||
|
raw_rows = all_raw_rows[:limit] if limit else all_raw_rows
|
||||||
|
print(f"Processing {len(raw_rows)} rows...")
|
||||||
|
|
||||||
|
# Build full PathCode map from all Excel rows for fallback resolution
|
||||||
|
excel_pathcode_html_map = {}
|
||||||
|
for vals in all_raw_rows:
|
||||||
|
pcode = str(vals[path_idx]).strip() if vals[path_idx] else ""
|
||||||
|
hval = vals[html_idx] or ""
|
||||||
|
if pcode:
|
||||||
|
excel_pathcode_html_map[pcode] = hval
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
pathcode_diag_map = {}
|
||||||
|
|
||||||
|
for vals in raw_rows:
|
||||||
|
chart_no = vals[chart_idx]
|
||||||
|
path_code = str(vals[path_idx]).strip() if vals[path_idx] else ""
|
||||||
|
html_content = vals[html_idx] or ""
|
||||||
|
|
||||||
|
tissue = extract_tissue(html_content, path_code)
|
||||||
|
diagnosis = extract_short_diagnosis(html_content, path_code)
|
||||||
|
|
||||||
|
site = classify_site(tissue, diagnosis, html_content)
|
||||||
|
ttype = classify_type(tissue, diagnosis, html_content)
|
||||||
|
|
||||||
|
row_dict = {
|
||||||
|
'chart_no': chart_no,
|
||||||
|
'path_code': path_code,
|
||||||
|
'site': site,
|
||||||
|
'type': ttype,
|
||||||
|
'tissue': tissue,
|
||||||
|
'diagnosis': diagnosis
|
||||||
|
}
|
||||||
|
rows.append(row_dict)
|
||||||
|
|
||||||
|
if path_code and diagnosis:
|
||||||
|
pathcode_diag_map[path_code] = diagnosis
|
||||||
|
|
||||||
|
print("Resolving empty diagnosis rows via 組織由來 (tissue) -> 病理號 (PathCode) matching...")
|
||||||
|
fallback_count = 0
|
||||||
|
for r in rows:
|
||||||
|
if not r['diagnosis'] and r['tissue']:
|
||||||
|
tissue = r['tissue'].strip()
|
||||||
|
potential_codes = [tissue] + re.findall(r'[A-Za-z0-9_\-]+', tissue)
|
||||||
|
for code in potential_codes:
|
||||||
|
if code not in pathcode_diag_map and code in excel_pathcode_html_map:
|
||||||
|
ref_html = excel_pathcode_html_map[code]
|
||||||
|
if ref_html:
|
||||||
|
db_diag = extract_short_diagnosis(ref_html, code)
|
||||||
|
if db_diag:
|
||||||
|
pathcode_diag_map[code] = db_diag
|
||||||
|
|
||||||
|
if code in pathcode_diag_map and pathcode_diag_map[code]:
|
||||||
|
r['diagnosis'] = pathcode_diag_map[code]
|
||||||
|
r['site'] = classify_site(r['tissue'], r['diagnosis'])
|
||||||
|
r['type'] = classify_type(r['tissue'], r['diagnosis'])
|
||||||
|
fallback_count += 1
|
||||||
|
break
|
||||||
|
|
||||||
|
print(f"Fallback resolution complete: updated {fallback_count} empty rows.")
|
||||||
|
print(f"Writing to CSV file: {output_csv} ...")
|
||||||
|
|
||||||
|
with open(output_csv, 'w', newline='', encoding='utf-8-sig') as f:
|
||||||
|
writer = csv.writer(f)
|
||||||
|
writer.writerow(['病歷號', '病理號', 'site', 'type', '組織由來', 'pathological diagnosis'])
|
||||||
|
|
||||||
|
for r in rows:
|
||||||
|
writer.writerow([r['chart_no'], r['path_code'], r['site'], r['type'], r['tissue'], r['diagnosis']])
|
||||||
|
|
||||||
|
print(f"Successfully exported {len(rows)} rows to {output_csv}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
inp = sys.argv[1] if len(sys.argv) > 1 else None
|
||||||
|
out = sys.argv[2] if len(sys.argv) > 2 else None
|
||||||
|
lim = int(sys.argv[3]) if len(sys.argv) > 3 and sys.argv[3].isdigit() else None
|
||||||
|
m = sys.argv[4] if len(sys.argv) > 4 else 'all'
|
||||||
|
process_extraction(inp, out, limit=lim, mode=m)
|
||||||
64
patho/0727/fetch.py
Executable file
64
patho/0727/fetch.py
Executable file
|
|
@ -0,0 +1,64 @@
|
||||||
|
#!/usr/bin/env python
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import re
|
||||||
|
|
||||||
|
# Add repository root and 'ntuh' directory to sys.path
|
||||||
|
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
base_dir = os.path.dirname(script_dir)
|
||||||
|
ntuh_dir = os.path.join(base_dir, 'ntuh')
|
||||||
|
|
||||||
|
if base_dir not in sys.path:
|
||||||
|
sys.path.insert(0, base_dir)
|
||||||
|
if ntuh_dir not in sys.path:
|
||||||
|
sys.path.insert(0, ntuh_dir)
|
||||||
|
|
||||||
|
os.environ.setdefault('DJANGO_SETTINGS_MODULE', 'ntuh.settings')
|
||||||
|
|
||||||
|
import django
|
||||||
|
django.setup()
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
from registry.models import PathologyReport
|
||||||
|
|
||||||
|
illegal_re = re.compile(r'[\x00-\x08\x0b-\x0c\x0e-\x1f]')
|
||||||
|
|
||||||
|
|
||||||
|
def clean_string_for_excel(val):
|
||||||
|
if not isinstance(val, str):
|
||||||
|
return val
|
||||||
|
# Remove illegal XML / openpyxl control characters
|
||||||
|
val = illegal_re.sub('', val)
|
||||||
|
# Strip unnecessary head, style, script, meta, link tags so report body fits within cell limits
|
||||||
|
h = re.sub(r'<(?:head|style|script|meta|link)[^>]*?>.*?</(?:head|style|script|meta|link)>', '', val, flags=re.DOTALL | re.IGNORECASE)
|
||||||
|
h = re.sub(r'<(?:meta|link)[^>]*?>', '', h, flags=re.IGNORECASE)
|
||||||
|
h = re.sub(r'\n\s*\n', '\n', h)
|
||||||
|
return h.strip()
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_pathology_reports(output_file=None):
|
||||||
|
if output_file is None:
|
||||||
|
output_file = os.path.join(script_dir, 'registry_pathologyreport.xlsx')
|
||||||
|
|
||||||
|
print(f"Fetching PathologyReport records from database...")
|
||||||
|
qs = PathologyReport.objects.all().values()
|
||||||
|
data = list(qs)
|
||||||
|
|
||||||
|
print(f"Cleaning string fields & HTML body for Excel cell limit...")
|
||||||
|
for item in data:
|
||||||
|
for k, v in item.items():
|
||||||
|
if isinstance(v, str):
|
||||||
|
item[k] = clean_string_for_excel(v)
|
||||||
|
|
||||||
|
df = pd.DataFrame(data)
|
||||||
|
print(f"Fetched {len(df)} rows. Writing to {output_file}...")
|
||||||
|
df.to_excel(output_file, index=False)
|
||||||
|
print(f"Successfully exported {len(df)} rows to {output_file}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
default_output = os.path.join(script_dir, 'registry_pathologyreport.xlsx')
|
||||||
|
target_path = sys.argv[1] if len(sys.argv) > 1 else default_output
|
||||||
|
fetch_pathology_reports(target_path)
|
||||||
266
patho/extract-medgemma.py
Normal file
266
patho/extract-medgemma.py
Normal file
|
|
@ -0,0 +1,266 @@
|
||||||
|
#!/usr/bin/env python
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import csv
|
||||||
|
import re
|
||||||
|
import json
|
||||||
|
import time
|
||||||
|
import datetime
|
||||||
|
import urllib.request
|
||||||
|
import urllib.parse
|
||||||
|
import openpyxl
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
|
|
||||||
|
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
base_dir = os.path.dirname(script_dir)
|
||||||
|
|
||||||
|
OLLAMA_URL = "http://192.168.221.20:11434/api/generate"
|
||||||
|
MODEL_NAME = "medgemma:27b"
|
||||||
|
LOG_FILE_PATH = os.path.join(script_dir, "ollama_calls.log")
|
||||||
|
|
||||||
|
|
||||||
|
def log_ollama_call(path_code, attempt, duration_sec, status, site="", ttype="", tissue="", diagnosis="", error=None):
|
||||||
|
timestamp = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||||
|
file_exists = os.path.exists(LOG_FILE_PATH) and os.path.getsize(LOG_FILE_PATH) > 0
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(LOG_FILE_PATH, "a", newline="", encoding="utf-8-sig") as f:
|
||||||
|
writer = csv.writer(f)
|
||||||
|
if not file_exists:
|
||||||
|
writer.writerow(["timestamp", "path_code", "attempt", "duration_sec", "status", "site", "type", "tissue", "diagnosis", "error"])
|
||||||
|
|
||||||
|
writer.writerow([
|
||||||
|
timestamp,
|
||||||
|
path_code,
|
||||||
|
attempt,
|
||||||
|
round(duration_sec, 3),
|
||||||
|
status,
|
||||||
|
site,
|
||||||
|
ttype,
|
||||||
|
tissue,
|
||||||
|
diagnosis,
|
||||||
|
str(error) if error else ""
|
||||||
|
])
|
||||||
|
except Exception as e:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def extract_plain_text(html_content):
|
||||||
|
if not html_content or not isinstance(html_content, str):
|
||||||
|
return ""
|
||||||
|
soup = BeautifulSoup(html_content, 'html.parser')
|
||||||
|
|
||||||
|
# Remove script and style tags
|
||||||
|
for s in soup(['script', 'style', 'head', 'meta', 'link']):
|
||||||
|
s.decompose()
|
||||||
|
|
||||||
|
text = soup.get_text(separator=' ').strip()
|
||||||
|
return re.sub(r'\s+', ' ', text)[:3000]
|
||||||
|
|
||||||
|
|
||||||
|
def enforce_site_rules(site, tissue, diagnosis):
|
||||||
|
tissue_lower = str(tissue).lower()
|
||||||
|
|
||||||
|
# Rule: If tissue is Nasopharynx or Nasal cavity, site is NOT cranial (should be other).
|
||||||
|
# However, sinonasal IS cranial.
|
||||||
|
if 'sinonasal' in tissue_lower:
|
||||||
|
site = 'cranial'
|
||||||
|
elif 'nasopharynx' in tissue_lower or 'nasal cavity' in tissue_lower:
|
||||||
|
if site == 'cranial':
|
||||||
|
site = 'other'
|
||||||
|
|
||||||
|
return site
|
||||||
|
|
||||||
|
|
||||||
|
def extract_with_medgemma_pure(html_content, path_code='', retries=5):
|
||||||
|
text_content = extract_plain_text(html_content)
|
||||||
|
|
||||||
|
prompt = f"""You are an expert medical pathology analysis assistant.
|
||||||
|
Analyze the following pathology report text for specimen {path_code} and extract key clinical information.
|
||||||
|
|
||||||
|
Report Text:
|
||||||
|
{text_content}
|
||||||
|
|
||||||
|
Tasks:
|
||||||
|
1. "site": Anatomical site classification. Must be EXACTLY one of: "cranial", "spinal", or "other".
|
||||||
|
- "cranial": Brain, cerebrum, cerebellum, brainstem, skull, cranial dura, pituitary, CP angle, orbit, sinonasal tissue, head, scalp, etc.
|
||||||
|
- "NOTE on Nasal cavity / Nasopharynx": If tissue origin is "Nasopharynx" or "Nasal cavity" alone, it is NOT cranial (classify as "other"). However, "sinonasal" IS cranial.
|
||||||
|
- "spinal": Spine, spinal cord, vertebra, lumbar/thoracic/cervical/sacral level, epidural, disc, ligamentum flavum, etc.
|
||||||
|
- "other": All other body organs or unspecified sites.
|
||||||
|
|
||||||
|
2. "type": Pathological nature. Must be EXACTLY one of: "tumor" or "other".
|
||||||
|
- "tumor": Neoplasms, carcinomas, adenomas, gliomas, meningiomas, schwannomas, metastases, lipomas, polyps, cysts, or neoplastic findings.
|
||||||
|
- "other": Non-neoplastic findings (e.g. inflammation, hematoma, abscess, necrosis, normal tissue, thrombus).
|
||||||
|
|
||||||
|
3. "tissue": Tissue origin / specimen site string in English (e.g., "Brain, cerebrum", "Soft tissue, L3", "Spine, L4-L5 level", "Bone, T12 vertebra", "Spinal cord, T7-T9").
|
||||||
|
- CRITICAL: Always keep and preserve the specific spine level if specified in the report (e.g. "L3", "L4/5", "T7-9", "C4-5", "T12", "S1-S2", "C1-C2"). Do not strip or omit spine levels!
|
||||||
|
|
||||||
|
4. "diagnosis": Concise histological / pathological diagnosis statement in English.
|
||||||
|
|
||||||
|
Return ONLY a valid JSON object matching this schema:
|
||||||
|
{{
|
||||||
|
"site": "cranial|spinal|other",
|
||||||
|
"type": "tumor|other",
|
||||||
|
"tissue": "tissue origin (including exact spine level if specified)",
|
||||||
|
"diagnosis": "pathological diagnosis"
|
||||||
|
}}
|
||||||
|
"""
|
||||||
|
|
||||||
|
payload = {
|
||||||
|
"model": MODEL_NAME,
|
||||||
|
"prompt": prompt,
|
||||||
|
"stream": False,
|
||||||
|
"format": "json"
|
||||||
|
}
|
||||||
|
|
||||||
|
for attempt in range(1, retries + 1):
|
||||||
|
start_time = time.time()
|
||||||
|
try:
|
||||||
|
req = urllib.request.Request(
|
||||||
|
OLLAMA_URL,
|
||||||
|
data=json.dumps(payload).encode('utf-8'),
|
||||||
|
headers={'Content-Type': 'application/json'}
|
||||||
|
)
|
||||||
|
with urllib.request.urlopen(req, timeout=90) as resp:
|
||||||
|
elapsed = time.time() - start_time
|
||||||
|
res_data = json.loads(resp.read().decode('utf-8'))
|
||||||
|
resp_text = res_data.get('response', '')
|
||||||
|
parsed = json.loads(resp_text)
|
||||||
|
|
||||||
|
site = str(parsed.get('site', '')).strip().lower()
|
||||||
|
ttype = str(parsed.get('type', '')).strip().lower()
|
||||||
|
tissue = str(parsed.get('tissue', '')).strip()
|
||||||
|
diagnosis = str(parsed.get('diagnosis', '')).strip()
|
||||||
|
|
||||||
|
if site not in ['cranial', 'spinal', 'other']:
|
||||||
|
site = 'other'
|
||||||
|
if ttype not in ['tumor', 'other']:
|
||||||
|
ttype = 'other'
|
||||||
|
|
||||||
|
site = enforce_site_rules(site, tissue, diagnosis)
|
||||||
|
|
||||||
|
log_ollama_call(
|
||||||
|
path_code=path_code,
|
||||||
|
attempt=attempt,
|
||||||
|
duration_sec=elapsed,
|
||||||
|
status="SUCCESS",
|
||||||
|
site=site,
|
||||||
|
ttype=ttype,
|
||||||
|
tissue=tissue,
|
||||||
|
diagnosis=diagnosis
|
||||||
|
)
|
||||||
|
|
||||||
|
return site, ttype, tissue, diagnosis
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
elapsed = time.time() - start_time
|
||||||
|
print(f"Warning: MedGemma API attempt {attempt}/{retries} error for {path_code}: {e}")
|
||||||
|
log_ollama_call(
|
||||||
|
path_code=path_code,
|
||||||
|
attempt=attempt,
|
||||||
|
duration_sec=elapsed,
|
||||||
|
status="ERROR",
|
||||||
|
error=e
|
||||||
|
)
|
||||||
|
if attempt < retries:
|
||||||
|
time.sleep(3 * attempt)
|
||||||
|
|
||||||
|
return "other", "other", "", ""
|
||||||
|
|
||||||
|
|
||||||
|
def process_single_row(args):
|
||||||
|
idx, chart_no, path_code, html_content, total_count = args
|
||||||
|
print(f"[{idx+1}/{total_count}] Extracting PathCode {path_code} with MedGemma...")
|
||||||
|
site, ttype, tissue, diagnosis = extract_with_medgemma_pure(html_content, path_code)
|
||||||
|
print(f" -> [{idx+1}/{total_count}] {path_code}: site={site}, type={ttype}, tissue={tissue!r}, diag={diagnosis!r}")
|
||||||
|
return idx, [chart_no, path_code, site, ttype, tissue, diagnosis]
|
||||||
|
|
||||||
|
|
||||||
|
def process_extraction_medgemma(input_excel=None, output_csv=None, limit=0, prefix_filter=None, max_workers=4):
|
||||||
|
if input_excel is None:
|
||||||
|
input_excel = os.path.join(script_dir, 'registry_pathologyreport.xlsx')
|
||||||
|
if not os.path.exists(input_excel) and os.path.exists('registry_pathologyreport.xlsx'):
|
||||||
|
input_excel = 'registry_pathologyreport.xlsx'
|
||||||
|
|
||||||
|
if output_csv is None:
|
||||||
|
if prefix_filter:
|
||||||
|
output_csv = os.path.join(script_dir, f'registry_pathologyreport_medgemma_{prefix_filter}.csv')
|
||||||
|
else:
|
||||||
|
output_csv = os.path.join(script_dir, 'registry_pathologyreport_medgemma_all.csv')
|
||||||
|
|
||||||
|
print(f"Opening Excel file: {input_excel} ...")
|
||||||
|
wb = openpyxl.load_workbook(input_excel, read_only=True)
|
||||||
|
ws = wb.active
|
||||||
|
|
||||||
|
headers = None
|
||||||
|
chart_idx = None
|
||||||
|
path_idx = None
|
||||||
|
html_idx = None
|
||||||
|
|
||||||
|
raw_rows = []
|
||||||
|
print(f"Reading Excel rows (prefix filter: {prefix_filter!r}, limit: {limit})...")
|
||||||
|
for i, row in enumerate(ws.iter_rows()):
|
||||||
|
vals = [cell.value for cell in row]
|
||||||
|
if headers is None:
|
||||||
|
headers = vals
|
||||||
|
chart_idx = headers.index('ChartNo')
|
||||||
|
path_idx = headers.index('PathCode')
|
||||||
|
html_idx = headers.index('html')
|
||||||
|
continue
|
||||||
|
|
||||||
|
path_code = str(vals[path_idx]).strip() if vals[path_idx] else ""
|
||||||
|
if prefix_filter:
|
||||||
|
if not path_code.startswith(prefix_filter):
|
||||||
|
continue
|
||||||
|
|
||||||
|
raw_rows.append(vals)
|
||||||
|
if limit and len(raw_rows) >= limit:
|
||||||
|
break
|
||||||
|
|
||||||
|
total_count = len(raw_rows)
|
||||||
|
print(f"Total rows to extract: {total_count} (max_workers: {max_workers})...")
|
||||||
|
|
||||||
|
tasks_args = [
|
||||||
|
(idx, vals[chart_idx], str(vals[path_idx]).strip() if vals[path_idx] else "", vals[html_idx] or "", total_count)
|
||||||
|
for idx, vals in enumerate(raw_rows)
|
||||||
|
]
|
||||||
|
|
||||||
|
results_dict = {}
|
||||||
|
completed_count = 0
|
||||||
|
|
||||||
|
# Write CSV header initially
|
||||||
|
with open(output_csv, 'w', newline='', encoding='utf-8-sig') as f:
|
||||||
|
writer = csv.writer(f)
|
||||||
|
writer.writerow(['病歷號', '病理號', 'site', 'type', '組織由來', 'pathological diagnosis'])
|
||||||
|
|
||||||
|
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||||
|
future_to_idx = {executor.submit(process_single_row, arg): arg[0] for arg in tasks_args}
|
||||||
|
|
||||||
|
for future in as_completed(future_to_idx):
|
||||||
|
idx, res_row = future.result()
|
||||||
|
results_dict[idx] = res_row
|
||||||
|
completed_count += 1
|
||||||
|
|
||||||
|
# Periodically dump progress every 50 rows
|
||||||
|
if completed_count % 50 == 0 or completed_count == total_count:
|
||||||
|
print(f"Progress: {completed_count}/{total_count} rows completed ({completed_count/total_count*100:.2f}%)...")
|
||||||
|
|
||||||
|
print(f"Writing all {total_count} rows in original order to CSV: {output_csv} ...")
|
||||||
|
with open(output_csv, 'w', newline='', encoding='utf-8-sig') as f:
|
||||||
|
writer = csv.writer(f)
|
||||||
|
writer.writerow(['病歷號', '病理號', 'site', 'type', '組織由來', 'pathological diagnosis'])
|
||||||
|
for idx in range(total_count):
|
||||||
|
writer.writerow(results_dict[idx])
|
||||||
|
|
||||||
|
print(f"Successfully exported {total_count} rows to {output_csv}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
lim = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else 0
|
||||||
|
pref = sys.argv[2] if len(sys.argv) > 2 and sys.argv[2] != 'None' else None
|
||||||
|
out = sys.argv[3] if len(sys.argv) > 3 else None
|
||||||
|
workers = int(sys.argv[4]) if len(sys.argv) > 4 and sys.argv[4].isdigit() else 4
|
||||||
|
process_extraction_medgemma(limit=lim, prefix_filter=pref, output_csv=out, max_workers=workers)
|
||||||
423
patho/extract.py
Executable file
423
patho/extract.py
Executable file
|
|
@ -0,0 +1,423 @@
|
||||||
|
#!/usr/bin/env python
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import csv
|
||||||
|
import re
|
||||||
|
import openpyxl
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
base_dir = os.path.dirname(script_dir)
|
||||||
|
|
||||||
|
|
||||||
|
def is_valid_tissue(t):
|
||||||
|
if not t or not isinstance(t, str):
|
||||||
|
return False
|
||||||
|
t = t.strip()
|
||||||
|
clean_t = re.sub(r'[\.,\-\_\s\(\)\"\']', '', t)
|
||||||
|
clean_no_dim = re.sub(r'\d+[\.\*x\d]*cm', '', clean_t, flags=re.IGNORECASE)
|
||||||
|
if not clean_t or len(clean_t) <= 2 or clean_t.lower() in ['none', 'nil']:
|
||||||
|
return False
|
||||||
|
if clean_t.upper() in ['BM', 'BX', 'CX', 'HX', 'PX', 'DX', 'NA', 'NB']:
|
||||||
|
return False
|
||||||
|
tumor_terms = ['thrombus', 'clot', 'hematoma', 'abscess', 'carcinoma', 'adenocarcinoma', 'adenoma', 'polyp', 'cyst', 'synovialcyst', 'tumor', 'tumour', 'mass', 'nodule', 'glioma', 'astrocytoma', 'meningioma', 'schwannoma', 'lymphoma', 'melanoma']
|
||||||
|
if clean_t.lower() in tumor_terms or any(clean_t.lower().endswith(k) for k in ['cyst', 'tumor', 'tumour', 'glioma']) or any(clean_no_dim.lower().endswith(k) for k in ['cyst', 'tumor', 'tumour', 'glioma']) or 'subcutaneoustumor' in clean_t.lower() or 'intraduraltumor' in clean_t.lower() or 'hyperintensity' in clean_t.lower() or 'hypointensity' in clean_t.lower() or 'duraseeding' in clean_t.lower() or 'cavitytumor' in clean_t.lower():
|
||||||
|
return False
|
||||||
|
if '位置不明' in t or 'sitenotstate' in clean_t.lower() or 'siteunspecified' in clean_t.lower():
|
||||||
|
return False
|
||||||
|
if any(k in clean_t.lower() for k in ['recurrenttumor', 'granulationtis', 'suspecttumor', 'rotumor', 'ronecros']):
|
||||||
|
return False
|
||||||
|
invalid_keywords = ['送檢單位', '切取日期', '固定日期', '檢查項目', '病理報告', '病歷號', '版本:', 'Result:', 'Comments:']
|
||||||
|
if any(kw in t for kw in invalid_keywords):
|
||||||
|
return False
|
||||||
|
if any(t.startswith(prefix) for prefix in ['送檢', '切取', '檢查', '病理', '病歷', '版本']):
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def extract_tissue(html_content, path_code=''):
|
||||||
|
if not html_content or not isinstance(html_content, str):
|
||||||
|
return ""
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html_content, 'html.parser')
|
||||||
|
|
||||||
|
tissue = ""
|
||||||
|
# 1. Check lblTissue span
|
||||||
|
span = soup.find('span', id=lambda x: x and 'lblTissue' in x)
|
||||||
|
if span and is_valid_tissue(span.text):
|
||||||
|
tissue = span.text.strip()
|
||||||
|
|
||||||
|
# 2. Check table row with 組織由來
|
||||||
|
if not tissue:
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
ths = [th.text.strip() for th in tr.find_all('th', recursive=False)]
|
||||||
|
tds = [td.text.strip() for td in tr.find_all('td', recursive=False)]
|
||||||
|
if '組織由來' in ths:
|
||||||
|
idx = ths.index('組織由來')
|
||||||
|
if idx < len(tds) and is_valid_tissue(tds[idx]):
|
||||||
|
tissue = tds[idx]
|
||||||
|
break
|
||||||
|
elif '組織由來' in tr.text:
|
||||||
|
if tds and is_valid_tissue(tds[-1]):
|
||||||
|
tissue = tds[-1]
|
||||||
|
|
||||||
|
# 3. Check 器官與術式 / 器官 / 部位 / 檢體部位
|
||||||
|
if not tissue:
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
ths = [th.text.strip() for th in tr.find_all('th', recursive=False)]
|
||||||
|
tds = [td.text.strip() for td in tr.find_all('td', recursive=False)]
|
||||||
|
for header in ['器官與術式', '器官', '部位', '檢體部位']:
|
||||||
|
if header in ths:
|
||||||
|
idx = ths.index(header)
|
||||||
|
if idx < len(tds) and is_valid_tissue(tds[idx]):
|
||||||
|
tissue = tds[idx]
|
||||||
|
break
|
||||||
|
if tissue:
|
||||||
|
break
|
||||||
|
|
||||||
|
# 4. Dig site from 組織學診斷 row if tissue is still empty
|
||||||
|
if not tissue:
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
ths = [th.text.strip() for th in tr.find_all('th', recursive=False)]
|
||||||
|
tds = [td.text.strip() for td in tr.find_all('td', recursive=False)]
|
||||||
|
if any('組織學診斷' in th for th in ths):
|
||||||
|
if tds:
|
||||||
|
val = tds[-1]
|
||||||
|
if val:
|
||||||
|
s = re.sub(r'^\d+[\.\s]*', '', val)
|
||||||
|
split_site = re.split(r',\s*(?:\d+(?:\-\d+)?\s*cm(?:\s+AAV|\s+above\s+anal\s+verge|\s+[A-Za-z]+)?|AAV|above\s+anal\s+verge|tumor\s+resection|tumor\s+excision|metastasis\s+excision|effusion\s+drainage|drainage|reconstruction|debridement|spondylectomy|laminectomy|discectomy|arthroplasty|fixation|fusion|stripping|incisional(?:\s+biopsy)?|excisional(?:\s+biopsy)?|punch(?:\s+biopsy)?|shave(?:\s+biopsy)?|wedge(?:\s+biopsy)?|core(?:\s+biopsy)?|site\s+not\s+state(?:d)?|site\s+unspecified|\?\s*side|\?|craniotomy(?:\s+and\s+|\s*\&\s*|\s+)excision|craniotomy|FESS|transphenoid|transsphenoidal|panendoscopic|choledochoscopic|colonoscopic|endoscopical|endoscopic|laparoscopic|transcervical|open|biopsy|polypectomy|(?:partial|subtotal|radical|wide|total)?\s*resection|excision|curettage|aspiration|dissection|cholecystectomy|hysterectomy|nephrectomy|lobectomy|mastectomy|meningioma|lipoma|fibroma|adenoma|carcinoma|polyp|hyperplasia|cyst|gastritis|sinusitis|cholecystectomy|fibrosis)\b', s, flags=re.IGNORECASE)[0]
|
||||||
|
split_site = re.split(r',\s*(?:corpus|cervix)\b', split_site, flags=re.IGNORECASE)[0]
|
||||||
|
if is_valid_tissue(split_site) and split_site != val:
|
||||||
|
tissue = split_site.strip()
|
||||||
|
break
|
||||||
|
|
||||||
|
# 5. Check active report tab link: <a href="...contentHolder" title="...">
|
||||||
|
if not tissue:
|
||||||
|
a_tag = soup.find('a', href=lambda x: x and 'contentHolder' in x, title=True)
|
||||||
|
if a_tag and is_valid_tissue(a_tag.get('title')):
|
||||||
|
tissue = a_tag['title'].strip()
|
||||||
|
|
||||||
|
# 6. Check selected option: <option selected="selected">... (tissue)</option>
|
||||||
|
if not tissue:
|
||||||
|
opt = soup.find('option', selected=True)
|
||||||
|
if opt and opt.text:
|
||||||
|
m = re.search(r'\(([^()]+)\)$', opt.text)
|
||||||
|
if m and is_valid_tissue(m.group(1)):
|
||||||
|
tissue = m.group(1).strip()
|
||||||
|
|
||||||
|
# 7. Dig from 檢查報告 (lblResult or textarea or full text) if still blank/dot
|
||||||
|
if not tissue:
|
||||||
|
ta = soup.find('textarea')
|
||||||
|
res_span = soup.find('span', id=lambda x: x and 'lblResult' in x)
|
||||||
|
raw_text = (ta.get_text(separator='\n') if ta else (res_span.get_text(separator='\n') if res_span else soup.get_text(separator='\n'))).strip()
|
||||||
|
if raw_text:
|
||||||
|
m = re.search(r'(?:Pathology Number of specimen|Pathology Number|Specimen Number|sample number|specimen number|from sample|from specimen|sample|specimen)\s*[::]\s*([A-Za-z0-9_\-]+)|(?:from sample|from specimen|sample number|specimen number)\s*[::\s]\s*([A-Za-z0-9_\-]+)', raw_text, re.IGNORECASE)
|
||||||
|
matched_code = (m.group(1) or m.group(2)).strip() if m else None
|
||||||
|
if matched_code:
|
||||||
|
tissue = matched_code
|
||||||
|
else:
|
||||||
|
lines = [l.strip() for l in raw_text.splitlines() if l.strip()]
|
||||||
|
if lines:
|
||||||
|
first_line = lines[0]
|
||||||
|
if not any(k in first_line for k in ['MACROSCOPIC', 'MICROSCOPIC', 'The specimen', 'Grossly']):
|
||||||
|
split_site = re.split(r',\s*(?:\d+(?:\-\d+)?\s*cm(?:\s+AAV|\s+above\s+anal\s+verge|\s+[A-Za-z]+)?|AAV|above\s+anal\s+verge)\b', first_line, flags=re.IGNORECASE)[0]
|
||||||
|
split_site = re.split(r',\s*(?:[A-Za-z\s]+)?(?:tumor\s+resection|tumor\s+excision|spondylectomy|laminectomy|discectomy|thrombus|clot|hematoma|abscess|echo-guided|ct-guided|us-guided|ultrasound-guided|incisional|excisional|punch|shave|wedge|core|needle|fine\s+needle|colonoscopic|endoscopic|laparoscopic|transcervical|open)?\s*(?:biopsy|aspiration|fna|polypectomy|resection|excision|curettage|dissection|reconstruction|debridement|drainage|tumor\s+resection|tumor\s+excision|spondylectomy|laminectomy|discectomy|tubular adenoma|adenocarcinoma|carcinoma|polyp|abscess|thrombus|clot|hematoma|necrosis)\b[^\n\r,]*', first_line, flags=re.IGNORECASE)[0]
|
||||||
|
split_site = re.split(r',\s*labeled\s+as\b', split_site, flags=re.IGNORECASE)[0]
|
||||||
|
if is_valid_tissue(split_site):
|
||||||
|
tissue = split_site.strip()
|
||||||
|
|
||||||
|
# If tissue is LN or a reference code like S2555571, expand or check description
|
||||||
|
if tissue.upper() == 'LN':
|
||||||
|
tissue = 'Lymph node'
|
||||||
|
|
||||||
|
if not tissue:
|
||||||
|
span = soup.find('span', id=lambda x: x and 'lblTissue' in x)
|
||||||
|
if span and span.text.strip().upper() == 'LN':
|
||||||
|
tissue = 'Lymph node'
|
||||||
|
|
||||||
|
if tissue and re.match(r'^[A-Za-z0-9_\-]+$', tissue):
|
||||||
|
ta = soup.find('textarea')
|
||||||
|
res_span = soup.find('span', id=lambda x: x and 'lblResult' in x)
|
||||||
|
raw_text = (ta.get_text(separator='\n') if ta else (res_span.get_text(separator='\n') if res_span else soup.get_text(separator='\n'))).strip()
|
||||||
|
if raw_text:
|
||||||
|
m = re.search(rf'{re.escape(tissue)}\s+([A-Za-z][^\n\r]+)', raw_text)
|
||||||
|
if m:
|
||||||
|
desc = m.group(1).strip()
|
||||||
|
split_site = re.split(r',\s*(?:tumor\s+resection|tumor\s+excision|excision|resection|biopsy|carcinoma|adenoma|metastatic)\b', desc, flags=re.IGNORECASE)[0]
|
||||||
|
if is_valid_tissue(split_site) and split_site != desc and not re.match(r'^[A-Za-z0-9_\-]+$', split_site):
|
||||||
|
tissue = split_site.strip()
|
||||||
|
|
||||||
|
return tissue
|
||||||
|
|
||||||
|
|
||||||
|
def extract_short_diagnosis(html_content, path_code=''):
|
||||||
|
if not html_content or not isinstance(html_content, str):
|
||||||
|
return ""
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html_content, 'html.parser')
|
||||||
|
|
||||||
|
# 1. Check table rows for 組織學診斷 / 診斷描述
|
||||||
|
table_diags = []
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
ths = [th.text.strip() for th in tr.find_all('th', recursive=False)]
|
||||||
|
tds = [td.text.strip() for td in tr.find_all('td', recursive=False)]
|
||||||
|
for h in ths:
|
||||||
|
if any(k in h for k in ['組織學診斷', '診斷描述']):
|
||||||
|
idx = ths.index(h)
|
||||||
|
if idx < len(tds):
|
||||||
|
val = tds[idx].text.strip() if hasattr(tds[idx], 'text') else str(tds[idx]).strip()
|
||||||
|
if val and val not in table_diags:
|
||||||
|
table_diags.append(val)
|
||||||
|
|
||||||
|
if not table_diags:
|
||||||
|
for tr in soup.find_all('tr'):
|
||||||
|
cells = [td.text.strip() for td in tr.find_all(['th', 'td'])]
|
||||||
|
text = ' '.join(cells)
|
||||||
|
if any(k in text for k in ['組織學診斷', '診斷描述']):
|
||||||
|
tds = tr.find_all('td')
|
||||||
|
if tds:
|
||||||
|
val = tds[-1].text.strip()
|
||||||
|
if val and val not in table_diags:
|
||||||
|
table_diags.append(val)
|
||||||
|
|
||||||
|
if table_diags:
|
||||||
|
return ', '.join(table_diags)
|
||||||
|
|
||||||
|
# 2. Check textarea or lblResult
|
||||||
|
ta = soup.find('textarea')
|
||||||
|
res_span = soup.find('span', id=lambda x: x and 'lblResult' in x)
|
||||||
|
raw_text = (ta.text if ta else (res_span.text if res_span else '')).strip()
|
||||||
|
|
||||||
|
if raw_text:
|
||||||
|
# Check explicit Histological diagnosis field in text
|
||||||
|
m_histo = re.search(r'Histological diagnosis[::\s]\s*([^\*\n\r]+)', raw_text, re.IGNORECASE)
|
||||||
|
if m_histo and m_histo.group(1).strip():
|
||||||
|
return m_histo.group(1).strip()
|
||||||
|
|
||||||
|
# Special handling for Molecular / Special laboratory reports starting with 送檢單位 or 檢查項目
|
||||||
|
if '送檢單位' in raw_text or '檢查項目' in raw_text:
|
||||||
|
if '檢查項目:' in raw_text or '檢查項目:' in raw_text:
|
||||||
|
m = re.search(r'檢查項目[::]\s*(.*?)(?:Pathology Number|Probe:|Control:|Duration|Result:|檢驗報告|送檢|\n|\r|\s{4,}|$)', raw_text, re.DOTALL)
|
||||||
|
if m and m.group(1).strip():
|
||||||
|
item = m.group(1).strip()
|
||||||
|
m_res = re.search(r'(?:Result:|檢驗報告[::])\s*(.*?)(?:Procedures:|Comments:|參考文獻|Ref:|\n\n|\r\n\r\n|\s{4,}|$)', raw_text, re.DOTALL)
|
||||||
|
if m_res and m_res.group(1).strip():
|
||||||
|
res_val = m_res.group(1).strip().replace('\n', ' ').replace('\t', ' ')
|
||||||
|
res_val = re.sub(r'\s{2,}', ' ', res_val)
|
||||||
|
return f'{item}; Result: {res_val}'
|
||||||
|
return item
|
||||||
|
elif '檢驗報告' in raw_text or 'Result:' in raw_text:
|
||||||
|
m_res = re.search(r'(?:Result:|檢驗報告[::])\s*(.*?)(?:Procedures:|Comments:|參考文獻|Ref:|\n\n|\r\n\r\n|\s{4,}|$)', raw_text, re.DOTALL)
|
||||||
|
if m_res and m_res.group(1).strip():
|
||||||
|
res_val = m_res.group(1).strip().replace('\n', ' ').replace('\t', ' ')
|
||||||
|
return re.sub(r'\s{2,}', ' ', res_val)
|
||||||
|
|
||||||
|
# Standard Pathology Report - return full first paragraph before macroscopic/gross section
|
||||||
|
cut_patterns = [
|
||||||
|
r'The specimen submitted', r'Grossly', r'Microscopically', r'Pathology Number',
|
||||||
|
r'送檢單位', r'檢查項目', r'Result:', r'Comments:', r'Ref:', r'\[\d+\]\.'
|
||||||
|
]
|
||||||
|
text_cut = re.split('|'.join(cut_patterns), raw_text, flags=re.IGNORECASE)[0]
|
||||||
|
parts = re.split(r'\n|\r|\t|\s{2,}', text_cut)
|
||||||
|
non_empty = [p.strip() for p in parts if p.strip()]
|
||||||
|
if non_empty:
|
||||||
|
return non_empty[0]
|
||||||
|
|
||||||
|
# 3. Fallback to lblBedDiagnosis if available
|
||||||
|
bed_span = soup.find('span', id=lambda x: x and 'lblBedDiagnosis' in x)
|
||||||
|
if bed_span and bed_span.text.strip():
|
||||||
|
return bed_span.text.strip()
|
||||||
|
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def classify_site(tissue, diagnosis, html=''):
|
||||||
|
site_text = (str(tissue) + ' ' + str(diagnosis)).lower()
|
||||||
|
site_text = re.sub(r'\bpd\-?l1\b', '', site_text, flags=re.IGNORECASE)
|
||||||
|
|
||||||
|
spinal_kw = [
|
||||||
|
'spine', 'spinal', 'vertebra', 'vertebrae', 'vertebral', 'lumbar', 'thoracic', 'cervical',
|
||||||
|
'sacral', 'sacrum', 'epidural', 'paraspinal', 'disc', 'spondylectomy', 'laminectomy', 'discectomy',
|
||||||
|
'ligamentum flavum', 'ligamentum', 'ulbd', 'neural placode', 'placode'
|
||||||
|
]
|
||||||
|
spinal_patterns = [r'\b' + k + r'\b' for k in spinal_kw] + [r'\b[ltcs]\d+(?:[\-\/]\d+)?\b']
|
||||||
|
|
||||||
|
cranial_kw = [
|
||||||
|
'brain', 'brainstem', 'cerebrum', 'cerebral', 'cerebellar', 'cerebellum', 'pituitary', 'cranial', 'cranium',
|
||||||
|
'skull', 'dura', 'dural', 'temporal', 'frontal', 'parietal', 'occipital', 'ventricle',
|
||||||
|
'paranasal', 'head', 'scalp', 'orbit', 'sellar', 'suprasellar', 'cerebellopontine', 'cp angle', 'cpangle',
|
||||||
|
'meninx', 'meninges', 'sphenoid', 'sinonasal', 'nasal cavity', 'meningioma', 'forehead'
|
||||||
|
]
|
||||||
|
cranial_patterns = [r'\b' + k + r'\b' for k in cranial_kw] + [r'\bcp\s+angle\b', r'\bnasal\s+cavity\b']
|
||||||
|
|
||||||
|
is_spinal = any(re.search(p, site_text) for p in spinal_patterns)
|
||||||
|
is_cranial = any(re.search(p, site_text) for p in cranial_patterns)
|
||||||
|
|
||||||
|
if 'dura mater' in site_text or 'dural' in site_text and 'epidural' not in site_text:
|
||||||
|
is_cranial = True
|
||||||
|
|
||||||
|
if is_spinal and not is_cranial:
|
||||||
|
return 'spinal'
|
||||||
|
elif is_cranial and not is_spinal:
|
||||||
|
return 'cranial'
|
||||||
|
elif is_spinal and is_cranial:
|
||||||
|
if any(k in site_text for k in ['brain', 'cerebrum', 'dura mater']):
|
||||||
|
return 'cranial'
|
||||||
|
return 'spinal'
|
||||||
|
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
|
||||||
|
def classify_type(tissue, diagnosis, html=''):
|
||||||
|
diag_str = str(diagnosis).lower()
|
||||||
|
text = (str(tissue) + ' ' + diag_str).lower()
|
||||||
|
|
||||||
|
if any(k in diag_str for k in ['no evidence of', 'negative for', 'free of', 'no tumor', 'without tumor', 'no residual']):
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
if any(k in diag_str for k in ['necrotic tissue', 'necrosis', 'organizing hematoma', 'hematoma', 'thrombus', 'clot', 'abscess', 'infarction', 'fibrosis', 'sinusitis', 'cholecystitis', 'gastritis', 'varicose']):
|
||||||
|
definitive_tumor = ['carcinoma', 'adenocarcinoma', 'adenoma', 'sarcoma', 'osteosarcoma', 'osteoma', 'astrocytoma', 'glioma', 'meningioma', 'glioblastoma', 'subependymoma', 'ependymoma', 'lipoma', 'fibroma', 'papilloma', 'lymphoma', 'melanoma', 'schwannoma', 'malignancy', 'malignant']
|
||||||
|
if not any(k in diag_str for k in definitive_tumor):
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
non_tumor_findings = ['granuloma', 'inflammatory cell infiltration', 'organizing hematoma', 'hematoma', 'thrombus', 'clot', 'abscess', 'necrosis', 'infarction', 'fibrosis', 'sinusitis', 'cholecystitis', 'gastritis', 'varicose', 'amyloid angiopathy']
|
||||||
|
neoplastic_findings = [
|
||||||
|
'carcinoma', 'adenocarcinoma', 'adenoma', 'sarcoma', 'osteosarcoma', 'osteoma', 'chondroma', 'hemangioma', 'neuroma', 'leiomyoma', 'myxoma',
|
||||||
|
'astrocytoma', 'glioma', 'meningioma', 'glioblastoma', 'subependymoma', 'ependymoma', 'ganglioglioma', 'oligodendroglioma', 'medulloblastoma', 'chordoma', 'hemangioblastoma', 'craniopharyngioma', 'lipoma', 'fibroma', 'papilloma',
|
||||||
|
'polyp', 'hyperplasia', 'metastasis', 'metastatic', 'malignancy', 'malignant',
|
||||||
|
'schwannoma', 'neurofibroma', 'melanoma', 'lymphoma', 'plasmacytosis', 'neoplastic', 'who grade', 'cns who grade'
|
||||||
|
]
|
||||||
|
|
||||||
|
if any(k in diag_str for k in non_tumor_findings) and not any(k in diag_str for k in neoplastic_findings):
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
tumor_kw = [
|
||||||
|
'malignancy', 'malignant', 'osteoma', 'chondroma', 'hemangioma', 'neuroma', 'leiomyoma', 'myxoma',
|
||||||
|
'subependymoma', 'ependymoma', 'ganglioglioma', 'oligodendroglioma', 'medulloblastoma', 'chordoma', 'hemangioblastoma', 'craniopharyngioma', 'who grade', 'cns who grade', 'tumor', 'tumour', 'carcinoma', 'adenocarcinoma', 'adenoma', 'sarcoma', 'osteosarcoma',
|
||||||
|
'astrocytoma', 'glioma', 'meningioma', 'glioblastoma', 'lipoma', 'fibroma', 'papilloma', 'cyst',
|
||||||
|
'polyp', 'hyperplasia', 'mass', 'lesion', 'neoplasm', 'metastasis', 'metastatic',
|
||||||
|
'schwannoma', 'neurofibroma', 'melanoma', 'lymphoma', 'plasmacytosis', 'neoplastic'
|
||||||
|
]
|
||||||
|
|
||||||
|
if any(k in text for k in tumor_kw):
|
||||||
|
return 'tumor'
|
||||||
|
|
||||||
|
return 'other'
|
||||||
|
|
||||||
|
|
||||||
|
def process_extraction(input_excel=None, output_csv=None, limit=None, mode='all'):
|
||||||
|
if input_excel is None:
|
||||||
|
input_excel = os.path.join(script_dir, 'registry_pathologyreport.xlsx')
|
||||||
|
if not os.path.exists(input_excel) and os.path.exists('registry_pathologyreport.xlsx'):
|
||||||
|
input_excel = 'registry_pathologyreport.xlsx'
|
||||||
|
|
||||||
|
if output_csv is None:
|
||||||
|
output_csv = os.path.join(script_dir, 'registry_pathologyreport_extracted.csv')
|
||||||
|
|
||||||
|
print(f"Opening Excel file: {input_excel} ...")
|
||||||
|
wb = openpyxl.load_workbook(input_excel, read_only=True)
|
||||||
|
ws = wb.active
|
||||||
|
|
||||||
|
headers = None
|
||||||
|
chart_idx = None
|
||||||
|
path_idx = None
|
||||||
|
html_idx = None
|
||||||
|
|
||||||
|
all_raw_rows = []
|
||||||
|
|
||||||
|
print(f"Reading Excel rows...")
|
||||||
|
for i, row in enumerate(ws.iter_rows()):
|
||||||
|
vals = [cell.value for cell in row]
|
||||||
|
if headers is None:
|
||||||
|
headers = vals
|
||||||
|
chart_idx = headers.index('ChartNo')
|
||||||
|
path_idx = headers.index('PathCode')
|
||||||
|
html_idx = headers.index('html')
|
||||||
|
continue
|
||||||
|
|
||||||
|
all_raw_rows.append(vals)
|
||||||
|
|
||||||
|
if mode == 'last' and limit:
|
||||||
|
raw_rows = all_raw_rows[-limit:]
|
||||||
|
print(f"Processing last {len(raw_rows)} rows (out of {len(all_raw_rows)} total)...")
|
||||||
|
else:
|
||||||
|
raw_rows = all_raw_rows[:limit] if limit else all_raw_rows
|
||||||
|
print(f"Processing {len(raw_rows)} rows...")
|
||||||
|
|
||||||
|
# Build full PathCode map from all Excel rows for fallback resolution
|
||||||
|
excel_pathcode_html_map = {}
|
||||||
|
for vals in all_raw_rows:
|
||||||
|
pcode = str(vals[path_idx]).strip() if vals[path_idx] else ""
|
||||||
|
hval = vals[html_idx] or ""
|
||||||
|
if pcode:
|
||||||
|
excel_pathcode_html_map[pcode] = hval
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
pathcode_diag_map = {}
|
||||||
|
|
||||||
|
for vals in raw_rows:
|
||||||
|
chart_no = vals[chart_idx]
|
||||||
|
path_code = str(vals[path_idx]).strip() if vals[path_idx] else ""
|
||||||
|
html_content = vals[html_idx] or ""
|
||||||
|
|
||||||
|
tissue = extract_tissue(html_content, path_code)
|
||||||
|
diagnosis = extract_short_diagnosis(html_content, path_code)
|
||||||
|
|
||||||
|
site = classify_site(tissue, diagnosis, html_content)
|
||||||
|
ttype = classify_type(tissue, diagnosis, html_content)
|
||||||
|
|
||||||
|
row_dict = {
|
||||||
|
'chart_no': chart_no,
|
||||||
|
'path_code': path_code,
|
||||||
|
'site': site,
|
||||||
|
'type': ttype,
|
||||||
|
'tissue': tissue,
|
||||||
|
'diagnosis': diagnosis
|
||||||
|
}
|
||||||
|
rows.append(row_dict)
|
||||||
|
|
||||||
|
if path_code and diagnosis:
|
||||||
|
pathcode_diag_map[path_code] = diagnosis
|
||||||
|
|
||||||
|
print("Resolving empty diagnosis rows via 組織由來 (tissue) -> 病理號 (PathCode) matching...")
|
||||||
|
fallback_count = 0
|
||||||
|
for r in rows:
|
||||||
|
if not r['diagnosis'] and r['tissue']:
|
||||||
|
tissue = r['tissue'].strip()
|
||||||
|
potential_codes = [tissue] + re.findall(r'[A-Za-z0-9_\-]+', tissue)
|
||||||
|
for code in potential_codes:
|
||||||
|
if code not in pathcode_diag_map and code in excel_pathcode_html_map:
|
||||||
|
ref_html = excel_pathcode_html_map[code]
|
||||||
|
if ref_html:
|
||||||
|
db_diag = extract_short_diagnosis(ref_html, code)
|
||||||
|
if db_diag:
|
||||||
|
pathcode_diag_map[code] = db_diag
|
||||||
|
|
||||||
|
if code in pathcode_diag_map and pathcode_diag_map[code]:
|
||||||
|
r['diagnosis'] = pathcode_diag_map[code]
|
||||||
|
r['site'] = classify_site(r['tissue'], r['diagnosis'])
|
||||||
|
r['type'] = classify_type(r['tissue'], r['diagnosis'])
|
||||||
|
fallback_count += 1
|
||||||
|
break
|
||||||
|
|
||||||
|
print(f"Fallback resolution complete: updated {fallback_count} empty rows.")
|
||||||
|
print(f"Writing to CSV file: {output_csv} ...")
|
||||||
|
|
||||||
|
with open(output_csv, 'w', newline='', encoding='utf-8-sig') as f:
|
||||||
|
writer = csv.writer(f)
|
||||||
|
writer.writerow(['病歷號', '病理號', 'site', 'type', '組織由來', 'pathological diagnosis'])
|
||||||
|
|
||||||
|
for r in rows:
|
||||||
|
writer.writerow([r['chart_no'], r['path_code'], r['site'], r['type'], r['tissue'], r['diagnosis']])
|
||||||
|
|
||||||
|
print(f"Successfully exported {len(rows)} rows to {output_csv}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
inp = sys.argv[1] if len(sys.argv) > 1 else None
|
||||||
|
out = sys.argv[2] if len(sys.argv) > 2 else None
|
||||||
|
lim = int(sys.argv[3]) if len(sys.argv) > 3 and sys.argv[3].isdigit() else None
|
||||||
|
m = sys.argv[4] if len(sys.argv) > 4 else 'all'
|
||||||
|
process_extraction(inp, out, limit=lim, mode=m)
|
||||||
64
patho/fetch.py
Executable file
64
patho/fetch.py
Executable file
|
|
@ -0,0 +1,64 @@
|
||||||
|
#!/usr/bin/env python
|
||||||
|
# -*- coding: utf-8 -*-
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import re
|
||||||
|
|
||||||
|
# Add repository root and 'ntuh' directory to sys.path
|
||||||
|
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
base_dir = os.path.dirname(script_dir)
|
||||||
|
ntuh_dir = os.path.join(base_dir, 'ntuh')
|
||||||
|
|
||||||
|
if base_dir not in sys.path:
|
||||||
|
sys.path.insert(0, base_dir)
|
||||||
|
if ntuh_dir not in sys.path:
|
||||||
|
sys.path.insert(0, ntuh_dir)
|
||||||
|
|
||||||
|
os.environ.setdefault('DJANGO_SETTINGS_MODULE', 'ntuh.settings')
|
||||||
|
|
||||||
|
import django
|
||||||
|
django.setup()
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
from registry.models import PathologyReport
|
||||||
|
|
||||||
|
illegal_re = re.compile(r'[\x00-\x08\x0b-\x0c\x0e-\x1f]')
|
||||||
|
|
||||||
|
|
||||||
|
def clean_string_for_excel(val):
|
||||||
|
if not isinstance(val, str):
|
||||||
|
return val
|
||||||
|
# Remove illegal XML / openpyxl control characters
|
||||||
|
val = illegal_re.sub('', val)
|
||||||
|
# Strip unnecessary head, style, script, meta, link tags so report body fits within cell limits
|
||||||
|
h = re.sub(r'<(?:head|style|script|meta|link)[^>]*?>.*?</(?:head|style|script|meta|link)>', '', val, flags=re.DOTALL | re.IGNORECASE)
|
||||||
|
h = re.sub(r'<(?:meta|link)[^>]*?>', '', h, flags=re.IGNORECASE)
|
||||||
|
h = re.sub(r'\n\s*\n', '\n', h)
|
||||||
|
return h.strip()
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_pathology_reports(output_file=None):
|
||||||
|
if output_file is None:
|
||||||
|
output_file = os.path.join(script_dir, 'registry_pathologyreport.xlsx')
|
||||||
|
|
||||||
|
print(f"Fetching PathologyReport records from database...")
|
||||||
|
qs = PathologyReport.objects.all().values()
|
||||||
|
data = list(qs)
|
||||||
|
|
||||||
|
print(f"Cleaning string fields & HTML body for Excel cell limit...")
|
||||||
|
for item in data:
|
||||||
|
for k, v in item.items():
|
||||||
|
if isinstance(v, str):
|
||||||
|
item[k] = clean_string_for_excel(v)
|
||||||
|
|
||||||
|
df = pd.DataFrame(data)
|
||||||
|
print(f"Fetched {len(df)} rows. Writing to {output_file}...")
|
||||||
|
df.to_excel(output_file, index=False)
|
||||||
|
print(f"Successfully exported {len(df)} rows to {output_file}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
default_output = os.path.join(script_dir, 'registry_pathologyreport.xlsx')
|
||||||
|
target_path = sys.argv[1] if len(sys.argv) > 1 else default_output
|
||||||
|
fetch_pathology_reports(target_path)
|
||||||
Loading…
Reference in a new issue