|
import re
|
|
from datetime import datetime
|
|
from typing import Any, ClassVar
|
|
|
|
|
|
class ShippingBillParser:
|
|
"""
|
|
Dynamic, deterministic, non-AI Shipping Bill parser.
|
|
|
|
This parser is designed for the pdfplumber output produced by
|
|
PDFAnalyzer.
|
|
|
|
It intentionally does NOT depend on:
|
|
- sample PDF values
|
|
- fixed page coordinates
|
|
- PDF text-stream order
|
|
- exact section strings
|
|
|
|
It uses:
|
|
- visual words and their coordinates
|
|
- label anchors
|
|
- nearby visual rows/regions
|
|
- pdfplumber-detected tables
|
|
- field validation
|
|
- generic normalization
|
|
|
|
The parser returns:
|
|
{
|
|
"shipping_bill_json": {...},
|
|
"_trace": [...]
|
|
}
|
|
"""
|
|
|
|
PARENT_FIELDS: ClassVar[list[str]] = [
|
|
"doctype",
|
|
"port_code",
|
|
"state_of_origin",
|
|
"inv",
|
|
"cb_code",
|
|
"gwt",
|
|
"shipping_bill_no",
|
|
"port_of_loading",
|
|
"item",
|
|
"buyer_name",
|
|
"gwt_unit",
|
|
"shipping_bill_date",
|
|
"cntry_of_finaldstn",
|
|
"cntry_of_discharge",
|
|
"cont",
|
|
"pkg",
|
|
"iec",
|
|
"port_of_finaldstn",
|
|
"port_of_discharge",
|
|
"iec_branch_code",
|
|
"p_1a_mode",
|
|
"p_1a_meis",
|
|
"p_1a_dfrc",
|
|
"p_1a_assess",
|
|
"p_1a_dbk",
|
|
"p_1a_re_exp",
|
|
"p_1a_exam",
|
|
"p_1a_rodtp",
|
|
"p_1a_lut",
|
|
"p_1a_job",
|
|
"p_1a_licence",
|
|
"p_1b_exporter_name",
|
|
"p_1b_exporter_address",
|
|
"p_1b_cb_name",
|
|
"p_1b_dbk_ac_no",
|
|
"p_1b_consignee_name",
|
|
"p_1b_consignee_address",
|
|
"p_1b_aeo",
|
|
"p_1b_ifsc_no",
|
|
"p_1b_type",
|
|
"p_1b_rbi_wr_no",
|
|
"p_1b_gstin",
|
|
"p_1b_ad_code",
|
|
"p_1b_rbi_wr_dt",
|
|
"p_1b_forex_ac_no",
|
|
"p_1c_fob_val",
|
|
"p_1c_com",
|
|
"p_1c_cess",
|
|
"p_1c_freight",
|
|
"p_1c_deductions",
|
|
"p_1c_insurance",
|
|
"p_1c_p_c",
|
|
"p_1c_discount",
|
|
"p_1c_duty",
|
|
"p_1d_dbk",
|
|
"p_1d_rodtep_amt",
|
|
"p_1d_igst_amt",
|
|
"p_1d_rosctle_amt",
|
|
"p_1d_cess_amt",
|
|
"p_1d_igst_val",
|
|
"p_1j_subm_dt",
|
|
"p_1i_subm_time",
|
|
"p_1i_leo_no",
|
|
"p_1i_assess_dt",
|
|
"p_1i_assess_time",
|
|
"p_1i_leo_dt",
|
|
"p_1i_exmn_dt",
|
|
"p_1i_exmn_time",
|
|
"p_1i_leo_time",
|
|
"p_1i_brc_realzn_dt",
|
|
"p_1i_sez_unit_data",
|
|
]
|
|
|
|
CHILD_TABLES: ClassVar[list[str]] = [
|
|
"manifest_details",
|
|
"challan_details",
|
|
"annex_details",
|
|
"invoice_details",
|
|
"aa__dfia_licence_details",
|
|
"item_details",
|
|
"drawback_and_rosl_claim",
|
|
"job_details",
|
|
"single_window_declaration",
|
|
"single_window_declaration_constituents",
|
|
"single_window_declaration_control",
|
|
"supporting_documents",
|
|
"ar4_details",
|
|
"third_party_details",
|
|
"item_manufacturerproducergrower_details",
|
|
"rodtep_details",
|
|
"container_details",
|
|
]
|
|
|
|
COMPLETE_CHILD_FIELDS: ClassVar[dict[str, list[str]]] = {'manifest_details': ['p_1e_mawb_no', 'p_1e_noc', 'p_1e_mawb_dt', 'p_1e_cin_no', 'p_1e_hawb_no', 'p_1e_cin_dt', 'p_1e_hawb_dt', 'p_1e_cin_site_id'], 'challan_details': ['p_4h_challan_serial_no', 'p_4h_challan_no', 'p_4h_paymt_dt', 'p_4h_challan_amount'], 'annex_details': ['p_1i_seal_typ', 'p_1i_loose_pkts', 'p_1i_nature_cargo', 'p_1i_marks_numbers', 'p_1i_no_of_pkgs', 'p_1i_no_of_containers'], 'invoice_details': ['inv_sn', 'p_2a_inv_no', 'p_2a_inv_dt', 'p_2a_ad_code', 'p_2a_invterm', 'p_2b_exporter_name', 'p_2b_buyer_addr', 'p_2b_exporter_addr', 'p_2b_buyer_name', 'p_2c_invoice_value', 'p_2c_freight', 'p_2c_discount', 'p_2c_deduct', 'p_2c_exchng_rate_desc', 'p_2c_invoice_curr', 'p_2c_frieght_curr', 'p_2c_discount_curr', 'p_2c_deduct_curr', 'p_2c_fob_val', 'p_2c_insurance', 'p_2c_commison', 'p_2c_p_c', 'p_2c_fob_curr', 'p_2c_insurance_curr', 'p_2c_commison_curr', 'p_2c_p_c_curr'], 'aa__dfia_licence_details': ['p_4b_invsn', 'p_4b_itmsn', 'p_4b_licence_no', 'p_4b_licence_dt', 'p_4b_licence_value', 'p_4b_licence_qty'], 'item_details': ['p_3a_invsno', 'p_3a_itemsn', 'p_3a_cth', 'p_3a_item_desc', 'p_3a_qty', 'p_3a_uqc', 'p_3a_rate', 'p_3a_value', 'p_3a_fob', 'p_3a_pmv', 'p_3a_duty_amt', 'p_3a_cess_rate', 'p_3a_cess_amt', 'p_3a_dbk_claimed', 'p_3a_igststat', 'p_3a_igst_val', 'p_3a_igst_amt', 'p_3a_schcod', 'p_3a_scheme_desc', 'p_3a_sqc_mst', 'p_3a_sqc_uqc', 'p_3a_state_of_origin', 'p_3a_district_of_origin', 'p_3a_pt_abroad', 'p_3a_comp_cess', 'p_3a_end_use', 'p_3a_benefit_availd', 'p_3a_reward_benefit', 'p_3a_third_party_item'], 'drawback_and_rosl_claim': ['p_3b_invsno', 'p_3b_itemsn', 'p_3b_dbksn', 'p_3b_dbk_rate', 'p_3b_dbk_amt', 'p_3b_rosl_rate', 'p_3b_rosl_amt'], 'job_details': ['p_3c_invsn', 'p_3c_itmsn', 'p_3c_job_no', 'p_3c_job_date', 'p_3c_job_amount'], 'single_window_declaration': ['p_4d_invsn', 'p_4d_itmsn', 'p_4d_info', 'p_4d_qualifier', 'p_4d_info_cd', 'p_4d_info_text', 'p_4d_info_msr', 'p_4d_uqc'], 'single_window_declaration_constituents': ['p_4e_invsn', 'p_4e_itmsn', 'p_4e_constituent', 'p_4e_percentage'], 'single_window_declaration_control': ['p_4f_invsn', 'p_4f_itmsn', 'p_4f_control', 'p_4f_result'], 'supporting_documents': ['p_4g_invsn', 'p_4g_itmsn', 'p_4g_doc_code', 'p_4g_doc_no', 'p_4g_doc_date', 'p_4g_remarks'], 'ar4_details': ['p_4j_invsn', 'p_4j_itmsno', 'p_4j_ar4_number', 'p_4j_ar4_date', 'p_4j_commissionerate', 'p_4j_division', 'p_4j_range'], 'third_party_details': ['p_4k_invsn', 'p_4k_itmsn', 'p_4k_iec', 'p_4k_exporter_name', 'p_4k_address', 'p_4k_gstn_id_type'], 'item_manufacturerproducergrower_details': ['p_4l_invsn', 'p_4l_itmsn', 'p_4l_type', 'p_4l_manufact_cd', 'p_4l_source_state', 'p_4l_trans_cy', 'p_4l_address'], 'rodtep_details': ['p_4m_invsn', 'p_4m_itmsn', 'p_4m_quantity', 'p_4m_uqc', 'p_4m_no_of_units', 'p_4m_value'], 'container_details': ['p_4i_container', 'p_4i_cont_seal', 'p_4i_cont_dt', 'p_4i_cont_sno']}
|
|
|
|
# Empty child tables are represented by one schema-only row.
|
|
# The row contains canonical Frappe fieldnames only; it contains no
|
|
# invented PDF data.
|
|
EMIT_EMPTY_CHILD_SCHEMA_ROWS: ClassVar[bool] = True
|
|
|
|
CHILD_DOCTYPES: ClassVar[dict[str, str]] = {'manifest_details': 'Manifest Detail', 'challan_details': 'Challan Detail', 'annex_details': 'Annex Detail', 'invoice_details': 'Invoice Detail', 'aa__dfia_licence_details': 'AA or DFIA Licence Detail', 'item_details': 'Item Detail', 'drawback_and_rosl_claim': 'Drawback And Rosl Claim', 'job_details': 'Job Detail', 'single_window_declaration': 'Single Window Declaration', 'single_window_declaration_constituents': 'Single Window Declaration Constituent', 'single_window_declaration_control': 'Single Window Declaration Control', 'supporting_documents': 'Supporting Document', 'ar4_details': 'AR4 Detail', 'third_party_details': 'Third Party Detail', 'item_manufacturerproducergrower_details': 'Item Manufacturer-Producer-Grower Detail', 'rodtep_details': 'Rodtep Detail', 'container_details': 'Container Detail'}
|
|
|
|
STATUS_FIELDS: ClassVar[list[str]] = [
|
|
"p_1a_assess",
|
|
"p_1a_meis",
|
|
"p_1a_dbk",
|
|
"p_1a_re_exp",
|
|
"p_1a_exam",
|
|
"p_1a_rodtp",
|
|
"p_1a_lut",
|
|
"p_1a_job",
|
|
"p_1a_licence",
|
|
"p_1a_dfrc",
|
|
]
|
|
|
|
NUMBER_FIELDS: ClassVar[set[str]] = {'p_1c_fob_val', 'p_1c_discount', 'p_1d_dbk', 'cont', 'p_1d_rosctle_amt', 'p_1d_cess_amt', 'item', 'pkg', 'inv', 'p_1c_freight', 'p_1d_igst_val', 'p_1c_com', 'p_1c_deductions', 'p_1c_p_c', 'p_1d_rodtep_amt', 'p_1c_duty', 'p_1c_cess', 'p_1c_insurance', 'p_1d_igst_amt', 'gwt'}
|
|
|
|
DATE_FIELDS: ClassVar[set[str]] = {'p_1i_leo_dt', 'p_1i_assess_dt', 'p_1i_brc_realzn_dt', 'p_1i_exmn_dt', 'shipping_bill_date', 'p_1j_subm_dt'}
|
|
|
|
TIME_FIELDS: ClassVar[set[str]] = {'p_1i_subm_time', 'p_1i_assess_time', 'p_1i_exmn_time', 'p_1i_leo_time'}
|
|
|
|
LABELS: ClassVar[dict[str, list[str]]] = {
|
|
"port_code": ["PORT CODE"],
|
|
"shipping_bill_no": ["SB NO", "SB NO."],
|
|
"shipping_bill_date": ["SB DATE"],
|
|
"iec": ["IEC", "IEC/BR"],
|
|
"cb_code": ["CB CODE"],
|
|
"inv": ["INV"],
|
|
"item": ["ITEM"],
|
|
"cont": ["CONT"],
|
|
"pkg": ["PKG"],
|
|
"gwt": ["G.WT", "GWT", "GROSS WEIGHT"],
|
|
"gwt_unit": ["G.WT UNIT", "UNIT"],
|
|
"port_of_loading": ["PORT OF LOADING"],
|
|
"state_of_origin": ["STATE OF ORIGIN"],
|
|
"port_of_finaldstn": ["PORT OF FINAL DESTINATION"],
|
|
"port_of_discharge": ["PORT OF DISCHARGE"],
|
|
"cntry_of_finaldstn": [
|
|
"COUNTRY OF FINAL DESTINATION",
|
|
"COUNTRY OF FINALDESTINATION",
|
|
"COUNTRY OF FINALDESTINATIO",
|
|
],
|
|
"cntry_of_discharge": ["COUNTRY OF DISCHARGE"],
|
|
"p_1b_exporter_name": ["EXPORTER'S NAME & ADDRESS"],
|
|
"p_1b_consignee_name": ["CONSIGNEE NAME & ADDRESS"],
|
|
"p_1b_cb_name": ["CB NAME"],
|
|
"p_1b_type": ["TYPE"],
|
|
"p_1b_gstin": ["GSTIN", "GSTIN / TYPE"],
|
|
"p_1b_ad_code": ["AD CODE"],
|
|
"p_1b_forex_ac_no": ["FOREX BANK A/C NO", "FOREX BANK A/C NO."],
|
|
"p_1b_dbk_ac_no": ["DBK BANK A/C NO", "DBK BANK A/C NO."],
|
|
"p_1b_aeo": ["AEO"],
|
|
"p_1b_ifsc_no": ["IFSC NO", "IFSC NO."],
|
|
"p_1b_rbi_wr_no": ["RBI WAIVER NO", "RBI WAIVER NO."],
|
|
"p_1b_rbi_wr_dt": ["RBI WAIVER DT", "RBI WAIVER DATE"],
|
|
"p_1c_fob_val": ["FOB VALUE"],
|
|
"p_1c_cess": ["CESS", "9.CESS"],
|
|
"p_1c_duty": ["DUTY", "8.DUTY"],
|
|
"p_1c_freight": ["FREIGHT"],
|
|
"p_1c_discount": ["DISCOUNT"],
|
|
"p_1c_deductions": ["DEDUCTIONS", "DEDUCT"],
|
|
"p_1c_insurance": ["INSURANCE"],
|
|
"p_1c_com": ["COMMISSION", "COM"],
|
|
"p_1c_p_c": ["P & C", "P.C", "P C"],
|
|
"p_1d_dbk": ["DBK"],
|
|
"p_1d_rodtep_amt": ["RODTEP"],
|
|
"p_1d_rosctle_amt": ["ROSCTL", "ROSL"],
|
|
"p_1j_subm_dt": ["SUBMISSION"],
|
|
"p_1i_assess_dt": ["ASSESSMENT"],
|
|
"p_1i_assess_time": ["ASSESSMENT"],
|
|
"p_1i_exmn_dt": ["EXAMINATION"],
|
|
"p_1i_exmn_time": ["EXAMINATION"],
|
|
"p_1i_subm_time": ["SUBMISSION"],
|
|
"p_1i_leo_time": ["LEO TIME", "LEO"],
|
|
"p_1i_sez_unit_data": ["SEZ UNIT DETAILS", "SEZ UNIT"],
|
|
"p_1i_leo_no": ["LEO NO"],
|
|
"p_1i_leo_dt": ["LEO DATE"],
|
|
"p_1i_exmn_dt": ["EXAMINATION"],
|
|
"p_1i_brc_realzn_dt": ["BRC REALISATION DATE", "BRC REALIZATION DATE"],
|
|
}
|
|
|
|
TABLE_ALIASES: ClassVar[dict[str, list[str]]] = {
|
|
"manifest_details": ["MANIFEST", "MAWB", "CIN"],
|
|
"annex_details": ["ANNEX", "SEAL TYPE", "NATURE OF CARGO"],
|
|
"invoice_details": ["INVOICE", "INV NO", "INVOICE NO"],
|
|
"item_details": ["ITEM DETAILS", "HS CODE", "CTH", "ITEM DESCRIPTION"],
|
|
"single_window_declaration": [
|
|
"SINGLE WINDOW",
|
|
"QUALIFIER",
|
|
"INFO CODE",
|
|
"INFO",
|
|
],
|
|
"container_details": ["CONTAINER", "CONTAINER NO"],
|
|
"challan_details": ["CHALLAN"],
|
|
"aa__dfia_licence_details": ["AA", "DFIA", "LICENCE"],
|
|
"drawback_and_rosl_claim": ["DRAWBACK", "ROSL", "RODTEP"],
|
|
"job_details": ["JOB DETAILS"],
|
|
"supporting_documents": ["SUPPORTING DOCUMENT"],
|
|
"ar4_details": ["AR4"],
|
|
"third_party_details": ["THIRD PARTY"],
|
|
"item_manufacturerproducergrower_details": [
|
|
"MANUFACTURER",
|
|
"PRODUCER",
|
|
"GROWER",
|
|
],
|
|
"rodtep_details": ["RODTEP"],
|
|
"single_window_declaration_constituents": ["CONSTITUENT"],
|
|
"single_window_declaration_control": ["CONTROL", "CONTROL RESULT"],
|
|
"reexport_details": ["REEXPORT DETAILS", "BE SITE ID", "BE NUMBER"],
|
|
}
|
|
|
|
|
|
# =========================================================
|
|
# COMPLETE DYNAMIC FIELD-ALIAS CONTRACT
|
|
# =========================================================
|
|
# These aliases describe labels/headers used by Shipping Bill PDFs.
|
|
# They map PDF terminology to the EXISTING Frappe fieldnames. They do
|
|
# not add or change any DocType field.
|
|
PARENT_FIELD_ALIASES: ClassVar[dict[str, list[str]]] = {
|
|
"state_of_origin": ["STATE OF ORIGIN", "ORIGIN STATE"],
|
|
"buyer_name": ["BUYER NAME", "BUYER", "BUYER NAME & ADDRESS"],
|
|
"p_1a_mode": ["MODE", "MODE OF TRANSPORT"],
|
|
"p_1a_meis": ["MEIS"],
|
|
"p_1a_dfrc": ["DFRC"],
|
|
"p_1a_assess": ["ASSESS", "ASSESSMENT STATUS"],
|
|
"p_1a_dbk": ["DBK"],
|
|
"p_1a_re_exp": ["RE-EXP", "RE EXP", "REEXP", "RE-EXPORT"],
|
|
"p_1a_exam": ["EXMN", "EXAM", "EXAMINATION STATUS"],
|
|
"p_1a_rodtp": ["RODTP", "RODTEP"],
|
|
"p_1a_lut": ["LUT"],
|
|
"p_1a_job": ["JOBBING", "JOB"],
|
|
"p_1a_licence": ["LICENCE", "LICENSE"],
|
|
"p_1b_exporter_name": ["EXPORTER'S NAME & ADDRESS", "EXPORTER NAME & ADDRESS"],
|
|
"p_1b_cb_name": ["CB NAME", "CUSTOMS BROKER", "CUSTOMS BROKER NAME"],
|
|
"p_1b_dbk_ac_no": ["DBK BANK A/C NO", "DBK BANK A/C NO."],
|
|
"p_1b_aeo": ["AEO", "AEO NO"],
|
|
"p_1b_ifsc_no": ["IFSC NO", "IFSC NO."],
|
|
"p_1b_type": ["TYPE"],
|
|
"p_1b_rbi_wr_no": ["RBI WAIVER NO", "RBI WAIVER NO."],
|
|
"p_1b_gstin": ["GSTIN", "GSTIN / TYPE", "GSTN", "GSTN ID"],
|
|
"p_1b_ad_code": ["AD CODE", "ADCODE"],
|
|
"p_1b_rbi_wr_dt": ["RBI WAIVER DT", "RBI WAIVER DATE"],
|
|
"p_1b_forex_ac_no": ["FOREX BANK A/C NO", "FOREX BANK A/C NO."],
|
|
"p_1c_fob_val": ["FOB VALUE"],
|
|
"p_1c_com": ["COM", "COMMISSION", "COMMISSION VALUE"],
|
|
"p_1c_cess": ["CESS", "CESS AMOUNT"],
|
|
"p_1c_freight": ["FREIGHT"],
|
|
"p_1c_deductions": ["DEDUCTIONS", "DEDUCT"],
|
|
"p_1c_insurance": ["INSURANCE"],
|
|
"p_1c_p_c": ["P/C", "P & C", "P.C", "P C"],
|
|
"p_1c_discount": ["DISCOUNT"],
|
|
"p_1c_duty": ["DUTY", "DUTY AMOUNT"],
|
|
"p_1d_dbk": ["DBK CLAIM", "DBK"],
|
|
"p_1d_rodtep_amt": ["RODTEP AMT", "RODTEP AMOUNT"],
|
|
"p_1d_igst_amt": ["IGST AMT", "IGST AMOUNT"],
|
|
"p_1d_rosctle_amt": ["ROSCTL AMT", "ROSCTL AMOUNT", "ROSL AMT"],
|
|
"p_1d_cess_amt": ["CESS AMT", "CESS AMOUNT"],
|
|
"p_1d_igst_val": ["IGST VALUE", "IGST VAL"],
|
|
"p_1j_subm_dt": ["SUBMISSION", "SUBMISSION DATE"],
|
|
"p_1i_subm_time": ["SUBMISSION TIME"],
|
|
"p_1i_leo_no": ["LEO NO", "LEO NUMBER"],
|
|
"p_1i_assess_dt": ["ASSESSMENT DATE", "ASSESS DT"],
|
|
"p_1i_assess_time": ["ASSESSMENT TIME", "ASSESS TIME"],
|
|
"p_1i_leo_dt": ["LEO DATE", "LEO DT"],
|
|
"p_1i_exmn_dt": ["EXAMINATION DATE", "EXAMINATION DT", "EXMN DATE"],
|
|
"p_1i_exmn_time": ["EXAMINATION TIME", "EXMN TIME"],
|
|
"p_1i_leo_time": ["LEO TIME"],
|
|
"p_1i_brc_realzn_dt": ["BRC REALISATION DATE", "BRC REALIZATION DATE", "BRC REALZN DT"],
|
|
"p_1i_sez_unit_data": ["SEZ UNIT DETAILS", "SEZ UNIT DATA", "SEZ UNIT"],
|
|
}
|
|
|
|
CHILD_FIELD_ALIASES: ClassVar[dict[str, dict[str, list[str]]]] = {
|
|
"manifest_details": {
|
|
"p_1e_mawb_no": ["MAWB NO", "MAWB NUMBER", "MASTER AWB", "MASTER AWB NO"],
|
|
"p_1e_noc": ["N.O.C.", "NOC", "NO C", "NOC NO"],
|
|
"p_1e_mawb_dt": ["MAWB DT", "MAWB DATE", "MAWB DT"],
|
|
"p_1e_cin_no": ["CIN NO", "CIN NUMBER"],
|
|
"p_1e_hawb_no": ["HAWB NO", "HAWB NUMBER", "HOUSE AWB"],
|
|
"p_1e_cin_dt": ["CIN DT", "CIN DATE"],
|
|
"p_1e_hawb_dt": ["HAWB DT", "HAWB DATE"],
|
|
"p_1e_cin_site_id": ["CIN SITE ID", "SITE ID", "CIN SITE"],
|
|
},
|
|
"challan_details": {
|
|
"p_4h_challan_serial_no": ["CHALLAN SERIAL NO", "CHALLAN SR NO", "SERIAL NO", "SERIAL NUMBER"],
|
|
"p_4h_challan_no": ["CHALLAN NO", "CHALLAN NUMBER"],
|
|
"p_4h_paymt_dt": ["PAYMENT DT", "PAYMENT DATE", "PAYMT DT", "PAYMT DATE"],
|
|
"p_4h_challan_amount": ["CHALLAN AMOUNT", "AMOUNT", "PAYMENT AMOUNT"],
|
|
},
|
|
"annex_details": {
|
|
"p_1i_seal_typ": ["SEAL TYPE", "SEAL TYP", "SEAL"],
|
|
"p_1i_loose_pkts": ["LOOSE PKTS", "LOOSE PACKETS", "LOOSE PACKAGES"],
|
|
"p_1i_nature_cargo": ["NATURE OF CARGO", "NATURE CARGO", "CARGO NATURE"],
|
|
"p_1i_marks_numbers": ["MARKS NUMBERS", "MARKS & NUMBERS", "MARKS AND NUMBERS", "MARKS"],
|
|
"p_1i_no_of_pkgs": ["NO OF PKGS", "NO OF PACKAGES", "NO OF PACKETS", "PACKAGES", "PACKETS"],
|
|
"p_1i_no_of_containers": ["NO OF CONTAINERS", "NUMBER OF CONTAINERS", "CONTAINERS"],
|
|
},
|
|
"invoice_details": {
|
|
"inv_sn": ["INV SN", "INV SNO", "INVOICE SN", "INVOICE SERIAL NO"],
|
|
"p_2a_inv_no": ["INV NO", "INVOICE NO", "INVOICE NUMBER"],
|
|
"p_2a_inv_dt": ["INV DT", "INVOICE DATE", "INV DATE"],
|
|
"p_2a_ad_code": ["AD CODE", "ADCODE"],
|
|
"p_2a_invterm": ["INVTERM", "INV TERM", "INVOICE TERM", "INCO TERM", "INCOTERM"],
|
|
"p_2b_exporter_name": ["EXPORTER NAME", "EXPORTER'S NAME"],
|
|
"p_2b_buyer_addr": ["BUYER ADDRESS", "BUYER ADDR"],
|
|
"p_2b_exporter_addr": ["EXPORTER ADDRESS", "EXPORTER ADDR"],
|
|
"p_2b_buyer_name": ["BUYER NAME", "BUYER", "CONSIGNEE NAME"],
|
|
"p_2c_invoice_value": ["INVOICE VALUE", "INVOICE AMOUNT", "INV VALUE"],
|
|
"p_2c_freight": ["FREIGHT"],
|
|
"p_2c_discount": ["DISCOUNT"],
|
|
"p_2c_deduct": ["DEDUCT", "DEDUCTIONS"],
|
|
"p_2c_exchng_rate_desc": ["EXCHANGE RATE", "EXCHNG RATE", "EXCHANGE RATE DESC", "EXCHNG RATE DESC"],
|
|
"p_2c_invoice_curr": ["INVOICE CURR", "INVOICE CURRENCY", "INV CURRENCY", "CURRENCY"],
|
|
"p_2c_frieght_curr": ["FREIGHT CURR", "FREIGHT CURRENCY"],
|
|
"p_2c_discount_curr": ["DISCOUNT CURR", "DISCOUNT CURRENCY"],
|
|
"p_2c_deduct_curr": ["DEDUCT CURR", "DEDUCT CURRENCY"],
|
|
"p_2c_fob_val": ["FOB VALUE", "FOB VAL"],
|
|
"p_2c_insurance": ["INSURANCE"],
|
|
"p_2c_commison": ["COMMISSION", "COMMISON", "COMM"],
|
|
"p_2c_p_c": ["P/C", "P & C", "P.C"],
|
|
"p_2c_fob_curr": ["FOB CURR", "FOB CURRENCY"],
|
|
"p_2c_insurance_curr": ["INSURANCE CURR", "INSURANCE CURRENCY"],
|
|
"p_2c_commison_curr": ["COMMISSION CURR", "COMMISON CURR", "COMMISSION CURRENCY"],
|
|
"p_2c_p_c_curr": ["P/C CURR", "P & C CURR", "P C CURR"],
|
|
},
|
|
"aa__dfia_licence_details": {
|
|
"p_4b_invsn": ["INVSN", "INV SN", "INVOICE SN"],
|
|
"p_4b_itmsn": ["ITMSN", "ITEM SN", "ITEM SNO"],
|
|
"p_4b_licence_no": ["LICENCE NO", "LICENSE NO", "LICENCE NUMBER", "LICENSE NUMBER"],
|
|
"p_4b_licence_dt": ["LICENCE DT", "LICENCE DATE", "LICENSE DATE"],
|
|
"p_4b_licence_value": ["LICENCE VALUE", "LICENSE VALUE"],
|
|
"p_4b_licence_qty": ["LICENCE QTY", "LICENCE QUANTITY", "LICENSE QTY", "LICENSE QUANTITY"],
|
|
},
|
|
"item_details": {
|
|
"p_3a_invsno": ["INVSNO", "INV SNO", "INV SN", "INVOICE SNO"],
|
|
"p_3a_itemsn": ["ITEMSN", "ITEM SNO", "ITEM SN", "ITEM SERIAL NO"],
|
|
"p_3a_cth": ["CTH", "HS CODE", "HS CD", "HSN", "TARIFF CODE"],
|
|
"p_3a_item_desc": ["ITEM DESCRIPTION", "DESCRIPTION", "ITEM DESC", "DESCRIPTION OF GOODS"],
|
|
"p_3a_qty": ["QTY", "QUANTITY"],
|
|
"p_3a_uqc": ["UQC", "UNIT"],
|
|
"p_3a_rate": ["RATE"],
|
|
"p_3a_value": ["VALUE", "ITEM VALUE"],
|
|
"p_3a_fob": ["FOB"],
|
|
"p_3a_pmv": ["PMV"],
|
|
"p_3a_duty_amt": ["DUTY AMT", "DUTY AMOUNT"],
|
|
"p_3a_cess_rate": ["CESS RATE"],
|
|
"p_3a_cess_amt": ["CESS AMT", "CESS AMOUNT"],
|
|
"p_3a_dbk_claimed": ["DBK CLAIMED", "DBK"],
|
|
"p_3a_igststat": ["IGST STAT", "IGST STATUS", "IGST STAT."],
|
|
"p_3a_igst_val": ["IGST VALUE", "IGST VAL"],
|
|
"p_3a_igst_amt": ["IGST AMT", "IGST AMOUNT"],
|
|
"p_3a_schcod": ["SCHCOD", "SCHEME CODE", "SCH CODE"],
|
|
"p_3a_scheme_desc": ["SCHEME DESC", "SCHEME DESCRIPTION"],
|
|
"p_3a_sqc_mst": ["SQC MST", "SQC MASTER", "SQC"],
|
|
"p_3a_sqc_uqc": ["SQC UQC"],
|
|
"p_3a_state_of_origin": ["STATE OF ORIGIN"],
|
|
"p_3a_district_of_origin": ["DISTRICT OF ORIGIN", "DISTRICT"],
|
|
"p_3a_pt_abroad": ["PT ABROAD", "PORT ABROAD"],
|
|
"p_3a_comp_cess": ["COMP CESS", "COMPENSATION CESS"],
|
|
"p_3a_end_use": ["END USE", "END-USE"],
|
|
"p_3a_benefit_availd": ["BENEFIT AVAILD", "BENEFIT AVAILABLE", "BENEFIT AVAILED"],
|
|
"p_3a_reward_benefit": ["REWARD BENEFIT", "REWARD"],
|
|
"p_3a_third_party_item": ["THIRD PARTY ITEM", "THIRD PARTY"],
|
|
},
|
|
"drawback_and_rosl_claim": {
|
|
"p_3b_invsno": ["INVSNO", "INV SNO", "INV SN"],
|
|
"p_3b_itemsn": ["ITEMSN", "ITEM SNO", "ITEM SN"],
|
|
"p_3b_dbksn": ["DBKSN", "DBK SNO", "DBK SN"],
|
|
"p_3b_dbk_rate": ["DBK RATE"],
|
|
"p_3b_dbk_amt": ["DBK AMT", "DBK AMOUNT"],
|
|
"p_3b_rosl_rate": ["ROSL RATE", "ROSCTL RATE"],
|
|
"p_3b_rosl_amt": ["ROSL AMT", "ROSL AMOUNT", "ROSCTL AMT", "ROSCTL AMOUNT"],
|
|
},
|
|
"job_details": {
|
|
"p_3c_invsn": ["INVSN", "INV SN"],
|
|
"p_3c_itmsn": ["ITMSN", "ITEM SN"],
|
|
"p_3c_job_no": ["JOB NO", "JOB NUMBER"],
|
|
"p_3c_job_date": ["JOB DATE", "JOB DT"],
|
|
"p_3c_job_amount": ["JOB AMOUNT", "JOB VALUE", "AMOUNT"],
|
|
},
|
|
"single_window_declaration": {
|
|
"p_4d_invsn": ["INVSN", "INV SN", "INVOICE SN"],
|
|
"p_4d_itmsn": ["ITMSN", "ITEM SN", "ITEM SNO"],
|
|
"p_4d_info": ["INFO"],
|
|
"p_4d_qualifier": ["QUALIFIER"],
|
|
"p_4d_info_cd": ["INFO CODE", "INFO CD", "CODE"],
|
|
"p_4d_info_text": ["INFO TEXT", "TEXT"],
|
|
"p_4d_info_msr": ["INFO MSR", "MEASURE", "MEASUREMENT"],
|
|
"p_4d_uqc": ["UQC", "UNIT"],
|
|
},
|
|
"single_window_declaration_constituents": {
|
|
"p_4e_invsn": ["INVSN", "INV SN"],
|
|
"p_4e_itmsn": ["ITMSN", "ITEM SN"],
|
|
"p_4e_constituent": ["CONSTITUENT", "CONSTITUENTS", "CONSTITUENT NAME"],
|
|
"p_4e_percentage": ["PERCENTAGE", "PERCENT", "%"],
|
|
},
|
|
"single_window_declaration_control": {
|
|
"p_4f_invsn": ["INVSN", "INV SN"],
|
|
"p_4f_itmsn": ["ITMSN", "ITEM SN"],
|
|
"p_4f_control": ["CONTROL", "CONTROL TYPE"],
|
|
"p_4f_result": ["RESULT", "CONTROL RESULT"],
|
|
},
|
|
"supporting_documents": {
|
|
"p_4g_invsn": ["INVSN", "INV SN"],
|
|
"p_4g_itmsn": ["ITMSN", "ITEM SN"],
|
|
"p_4g_doc_code": ["DOC CODE", "DOCUMENT CODE"],
|
|
"p_4g_doc_no": ["DOC NO", "DOCUMENT NO", "DOCUMENT NUMBER"],
|
|
"p_4g_doc_date": ["DOC DATE", "DOCUMENT DATE"],
|
|
"p_4g_remarks": ["REMARKS", "REMARK"],
|
|
},
|
|
"ar4_details": {
|
|
"p_4j_invsn": ["INVSN", "INV SN"],
|
|
"p_4j_itmsno": ["ITMSNO", "ITEM SNO", "ITEM SN"],
|
|
"p_4j_ar4_number": ["AR4 NUMBER", "AR4 NO", "AR4 NO."],
|
|
"p_4j_ar4_date": ["AR4 DATE", "AR4 DT"],
|
|
"p_4j_commissionerate": ["COMMISSIONERATE", "COMMISSIONERATE NAME"],
|
|
"p_4j_division": ["DIVISION"],
|
|
"p_4j_range": ["RANGE"],
|
|
},
|
|
"third_party_details": {
|
|
"p_4k_invsn": ["INVSN", "INV SN"],
|
|
"p_4k_itmsn": ["ITMSN", "ITEM SN"],
|
|
"p_4k_iec": ["IEC", "IEC NO"],
|
|
"p_4k_exporter_name": ["EXPORTER NAME", "THIRD PARTY EXPORTER"],
|
|
"p_4k_address": ["ADDRESS", "EXPORTER ADDRESS"],
|
|
"p_4k_gstn_id_type": ["GSTN ID TYPE", "GSTIN ID TYPE", "GSTN TYPE"],
|
|
},
|
|
"item_manufacturerproducergrower_details": {
|
|
"p_4l_invsn": ["INVSN", "INV SN"],
|
|
"p_4l_itmsn": ["ITMSN", "ITEM SN"],
|
|
"p_4l_type": ["TYPE"],
|
|
"p_4l_manufact_cd": ["MANUFACT CD", "MANUFACTURER CODE", "MANUFACTURE CODE"],
|
|
"p_4l_source_state": ["SOURCE STATE", "STATE OF SOURCE"],
|
|
"p_4l_trans_cy": ["TRANS CY", "TRANSACTION COUNTRY", "TRANS COUNTRY"],
|
|
"p_4l_address": ["ADDRESS", "MANUFACTURER ADDRESS"],
|
|
},
|
|
"rodtep_details": {
|
|
"p_4m_invsn": ["INVSN", "INV SN"],
|
|
"p_4m_itmsn": ["ITMSN", "ITEM SN"],
|
|
"p_4m_quantity": ["QUANTITY", "QTY"],
|
|
"p_4m_uqc": ["UQC", "UNIT"],
|
|
"p_4m_no_of_units": ["NO OF UNITS", "NUMBER OF UNITS", "UNITS"],
|
|
"p_4m_value": ["VALUE", "RODTEP VALUE"],
|
|
},
|
|
"container_details": {
|
|
"p_4i_container": ["CONTAINER NO", "CONTAINER NUMBER", "CONTAINER"],
|
|
"p_4i_cont_seal": ["CONTAINER SEAL", "CONT SEAL", "SEAL NO", "SEAL NUMBER"],
|
|
"p_4i_cont_dt": ["CONTAINER DT", "CONTAINER DATE", "CONT DT", "CONT DATE"],
|
|
"p_4i_cont_sno": ["CONTAINER SNO", "CONTAINER S NO", "CONTAINER SR NO", "CONT SNO", "CONT S NO"],
|
|
},
|
|
}
|
|
|
|
def __init__(self, pages: list[dict[str, Any]]):
|
|
self.pages = pages or []
|
|
self.trace: list[dict[str, Any]] = []
|
|
|
|
# Build the complete visual word model once.
|
|
self.words = self._all_words()
|
|
|
|
# V29 compatibility layer:
|
|
# Several recovery methods operate page-by-page. The previous version
|
|
# referenced self.page_words without initializing it, which caused:
|
|
# AttributeError: 'ShippingBillParser' object has no attribute 'page_words'
|
|
#
|
|
# Keep a deterministic page-number -> words mapping.
|
|
self.page_words = {}
|
|
|
|
for word in self.words:
|
|
page_no = word.get("_page")
|
|
if page_no is None:
|
|
continue
|
|
|
|
self.page_words.setdefault(page_no, []).append(word)
|
|
|
|
for page_no in self.page_words:
|
|
self.page_words[page_no].sort(
|
|
key=lambda w: (
|
|
float(w.get("y0", 0)),
|
|
float(w.get("x0", 0)),
|
|
)
|
|
)
|
|
|
|
self.rows = self._build_rows(self.words)
|
|
self.tables = self._all_tables()
|
|
self.full_text = "\n".join(p.get("text", "") for p in self.pages)
|
|
|
|
# =========================================================
|
|
# FIELD / LABEL SAFETY
|
|
# =========================================================
|
|
|
|
KNOWN_LABELS: ClassVar[set[str]] = {
|
|
"PORT CODE",
|
|
"SB NO",
|
|
"SB NO.",
|
|
"SB DATE",
|
|
"IEC",
|
|
"IEC/BR",
|
|
"CB CODE",
|
|
"INV",
|
|
"ITEM",
|
|
"CONT",
|
|
"PKG",
|
|
"G.WT",
|
|
"GWT",
|
|
"GROSS WEIGHT",
|
|
"G.WT UNIT",
|
|
"UNIT",
|
|
"PORT OF LOADING",
|
|
"STATE OF ORIGIN",
|
|
"PORT OF FINAL DESTINATION",
|
|
"PORT OF DISCHARGE",
|
|
"COUNTRY OF FINAL DESTINATION",
|
|
"COUNTRY OF FINALDESTINATION",
|
|
"COUNTRY OF FINALDESTINATIO",
|
|
"COUNTRY OF DISCHARGE",
|
|
"EXPORTER'S NAME & ADDRESS",
|
|
"1.EXPORTER'S NAME & ADDRESS",
|
|
"EXPORTER NAME & ADDRESS",
|
|
"CONSIGNEE NAME & ADDRESS",
|
|
"7.CONSIGNEE NAME & ADDRESS",
|
|
"CB NAME",
|
|
"TYPE",
|
|
"GSTIN",
|
|
"GSTIN / TYPE",
|
|
"AD CODE",
|
|
"FOREX BANK A/C NO",
|
|
"FOREX BANK A/C NO.",
|
|
"FOB VALUE",
|
|
"FREIGHT",
|
|
"DISCOUNT",
|
|
"DEDUCTIONS",
|
|
"DEDUCT",
|
|
"INSURANCE",
|
|
"COMMISSION",
|
|
"COM",
|
|
"P & C",
|
|
"P.C",
|
|
"P C",
|
|
"DBK",
|
|
"RODTEP",
|
|
"ROSCTL",
|
|
"ROSL",
|
|
"SUBMISSION",
|
|
"LEO NO",
|
|
"LEO DATE",
|
|
"LEO TIME",
|
|
"EXAMINATION",
|
|
"BRC REALISATION DATE",
|
|
"BRC REALIZATION DATE",
|
|
"MODE",
|
|
"ASSESS",
|
|
"RE-EXP",
|
|
"RE EXP",
|
|
"REEXP",
|
|
"EXMN",
|
|
"EXAM",
|
|
"LUT",
|
|
"JOB",
|
|
"LICENCE",
|
|
"LICENSE",
|
|
"DFRC",
|
|
}
|
|
|
|
def _label_is_value(self, value):
|
|
"""
|
|
Reject a candidate when it is another PDF label.
|
|
|
|
The previous implementation could accept a nearby label as
|
|
the value of the preceding field. This is the main protection
|
|
against mappings such as:
|
|
|
|
Port Code -> SB No
|
|
SB No -> SB Date
|
|
SB Date -> INBOM4
|
|
"""
|
|
text = self._clean(value).upper().strip(" :.-")
|
|
if not text:
|
|
return True
|
|
|
|
norm = self._norm(text)
|
|
known = {self._norm(x) for x in self.KNOWN_LABELS}
|
|
|
|
if norm in known:
|
|
return True
|
|
|
|
# Numbered labels such as "12.PORT OF LOADING".
|
|
if re.match(
|
|
r"^\d+\s*[\.\)]\s*[A-Z]",
|
|
text,
|
|
):
|
|
return True
|
|
|
|
# Common section headings.
|
|
if any(
|
|
token in norm
|
|
for token in (
|
|
"shipping bill summary",
|
|
"invoice details",
|
|
"item details",
|
|
"export scheme details",
|
|
"single window declaration",
|
|
"declarations",
|
|
"manifest details",
|
|
"annex details",
|
|
)
|
|
):
|
|
return True
|
|
|
|
return False
|
|
|
|
def _same_visual_line_words(self, y, page_no, tolerance=3.5):
|
|
return [w for w in self.words if w["_page"] == page_no and abs(w["y0"] - y) <= tolerance]
|
|
|
|
def _row_after_label(self, label, max_gap=60):
|
|
"""
|
|
Return the first meaningful visual row below a label.
|
|
|
|
Rows are used instead of individual nearest words so that a
|
|
value containing multiple tokens remains intact.
|
|
"""
|
|
rows = [
|
|
row
|
|
for row in self.rows
|
|
if row["_page"] == label["_page"]
|
|
and row["_y"] >= label["y1"]
|
|
and row["_y"] - label["y1"] <= max_gap
|
|
]
|
|
|
|
rows.sort(key=lambda r: r["_y"])
|
|
|
|
for row in rows:
|
|
text = self._clean(row["text"])
|
|
if not text or self._label_is_value(text):
|
|
continue
|
|
|
|
return row
|
|
|
|
return None
|
|
|
|
def _value_from_header_column(
|
|
self,
|
|
label,
|
|
validator=None,
|
|
max_vertical_gap=65,
|
|
max_horizontal_distance=90,
|
|
):
|
|
"""
|
|
Read a header value from the visual column under a label.
|
|
|
|
Unlike the old nearest-word logic, another header label can
|
|
never be returned as the value.
|
|
"""
|
|
rows = [
|
|
row
|
|
for row in self.rows
|
|
if row["_page"] == label["_page"]
|
|
and row["_y"] >= label["y1"]
|
|
and row["_y"] - label["y1"] <= max_vertical_gap
|
|
]
|
|
|
|
label_center = (label["x0"] + label["x1"]) / 2
|
|
|
|
for row in sorted(rows, key=lambda r: r["_y"]):
|
|
candidates = []
|
|
|
|
for word in row["words"]:
|
|
center = (word["x0"] + word["x1"]) / 2
|
|
|
|
if abs(center - label_center) > max_horizontal_distance:
|
|
continue
|
|
|
|
value = self._clean(word["_text"])
|
|
|
|
if self._label_is_value(value):
|
|
continue
|
|
|
|
if validator and not validator(value):
|
|
continue
|
|
|
|
candidates.append(word)
|
|
|
|
if candidates:
|
|
candidates.sort(key=lambda w: abs(((w["x0"] + w["x1"]) / 2) - label_center))
|
|
return candidates[0]["_text"]
|
|
|
|
return None
|
|
|
|
def _value_right_same_row(
|
|
self,
|
|
label,
|
|
validator=None,
|
|
max_distance=180,
|
|
):
|
|
"""
|
|
Read a value to the right of a label on the same visual row.
|
|
"""
|
|
candidates = []
|
|
|
|
label_y = (label["y0"] + label["y1"]) / 2
|
|
|
|
for word in self.words:
|
|
if word["_page"] != label["_page"]:
|
|
continue
|
|
|
|
word_y = (word["y0"] + word["y1"]) / 2
|
|
|
|
if abs(word_y - label_y) > 4:
|
|
continue
|
|
|
|
if word["x0"] < label["x1"]:
|
|
continue
|
|
|
|
distance = word["x0"] - label["x1"]
|
|
|
|
if distance > max_distance:
|
|
continue
|
|
|
|
value = self._clean(word["_text"])
|
|
|
|
if self._label_is_value(value):
|
|
continue
|
|
|
|
if validator and not validator(value):
|
|
continue
|
|
|
|
candidates.append((distance, word))
|
|
|
|
if not candidates:
|
|
return None
|
|
|
|
candidates.sort(key=lambda item: item[0])
|
|
return candidates[0][1]["_text"]
|
|
|
|
def _value_below_column(
|
|
self,
|
|
label,
|
|
validator=None,
|
|
max_y_gap=80,
|
|
x_tolerance=120,
|
|
):
|
|
"""
|
|
Read a complete visual row below the label in the same column.
|
|
"""
|
|
label_center = (label["x0"] + label["x1"]) / 2
|
|
|
|
rows = [
|
|
row
|
|
for row in self.rows
|
|
if row["_page"] == label["_page"]
|
|
and row["_y"] >= label["y1"]
|
|
and row["_y"] - label["y1"] <= max_y_gap
|
|
]
|
|
|
|
for row in sorted(rows, key=lambda r: r["_y"]):
|
|
selected = []
|
|
|
|
for word in row["words"]:
|
|
center = (word["x0"] + word["x1"]) / 2
|
|
|
|
if abs(center - label_center) > x_tolerance:
|
|
continue
|
|
|
|
value = self._clean(word["_text"])
|
|
|
|
if self._label_is_value(value):
|
|
continue
|
|
|
|
if validator and not validator(value):
|
|
continue
|
|
|
|
selected.append(word)
|
|
|
|
if selected:
|
|
selected.sort(key=lambda w: w["x0"])
|
|
return " ".join(w["_text"] for w in selected)
|
|
|
|
return None
|
|
|
|
# =========================================================
|
|
# ENTRY POINT
|
|
# =========================================================
|
|
|
|
# =========================================================
|
|
# FINAL REFERENCE-PDF GROSS WEIGHT RECOVERY
|
|
# =========================================================
|
|
|
|
def _force_recover_reference_gross_weight(self, result):
|
|
# Never overwrite an already valid extraction.
|
|
try:
|
|
existing_gwt = float(result.get("gwt", 0) or 0)
|
|
except (TypeError, ValueError):
|
|
existing_gwt = 0.0
|
|
try:
|
|
existing_pkg = int(float(result.get("pkg", 0) or 0))
|
|
except (TypeError, ValueError):
|
|
existing_pkg = 0
|
|
if existing_gwt > 0 and existing_pkg > 0:
|
|
if not result.get("gwt_unit"):
|
|
result["gwt_unit"] = "KGS"
|
|
return True
|
|
|
|
text = str(self.full_text or "")
|
|
lines = [self._clean(x) for x in text.splitlines() if self._clean(x)]
|
|
|
|
# Common compact form: PKG <count> G.WT KGS <weight>.
|
|
for line in lines:
|
|
pkg_match = re.search(r"\bPKG\s+(\d{1,5})\b", line, re.I)
|
|
wt_match = re.search(
|
|
r"G\s*\.?\s*W\s*\.?\s*T\s+(KG|KGS)\s+"
|
|
r"(\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d+(?:\.\d+)?)",
|
|
line,
|
|
re.I,
|
|
)
|
|
if wt_match:
|
|
try:
|
|
gwt = float(wt_match.group(2).replace(",", ""))
|
|
except (TypeError, ValueError):
|
|
continue
|
|
if 0 < gwt <= 10000000:
|
|
result["gwt"] = gwt
|
|
result["gwt_unit"] = wt_match.group(1).upper()
|
|
if pkg_match:
|
|
result["pkg"] = int(pkg_match.group(1))
|
|
return True
|
|
|
|
# Generic count + unit + weight pattern.
|
|
pattern = re.compile(
|
|
r"(?<!\d)(\d{1,5})\s+(KG|KGS)\s+"
|
|
r"(\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d+(?:\.\d+)?)(?!\d)",
|
|
re.I,
|
|
)
|
|
for line in lines:
|
|
match = pattern.search(line)
|
|
if not match:
|
|
continue
|
|
try:
|
|
pkg = int(match.group(1))
|
|
gwt = float(match.group(3).replace(",", ""))
|
|
except (TypeError, ValueError):
|
|
continue
|
|
if 0 <= pkg <= 100000 and 0 < gwt <= 10000000:
|
|
result["pkg"] = pkg
|
|
result["gwt"] = gwt
|
|
result["gwt_unit"] = match.group(2).upper()
|
|
return True
|
|
|
|
# Explicit G.WT label with a value in the same or next layout line.
|
|
for i, line in enumerate(lines):
|
|
if not re.search(r"G\s*\.?\s*W\s*\.?\s*T|GWT|GROSS\s+WEIGHT", line, re.I):
|
|
continue
|
|
window = line + (" " + lines[i + 1] if i + 1 < len(lines) else "")
|
|
match = re.search(
|
|
r"(\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d+(?:\.\d+)?)\s*(KG|KGS)\b",
|
|
window,
|
|
re.I,
|
|
)
|
|
if match:
|
|
try:
|
|
gwt = float(match.group(1).replace(",", ""))
|
|
except (TypeError, ValueError):
|
|
continue
|
|
if 0 < gwt <= 10000000:
|
|
result["gwt"] = gwt
|
|
result["gwt_unit"] = match.group(2).upper()
|
|
return True
|
|
|
|
return False
|
|
|
|
def _normalize_manifest_date(self, value):
|
|
"""
|
|
Normalize manifest/CIN dates to YYYY-MM-DD.
|
|
|
|
Accepts the common customs format DD-MMM-YY / DD-MMM-YYYY and
|
|
already-normalized ISO dates.
|
|
"""
|
|
value = self._clean(value)
|
|
if not value:
|
|
return ""
|
|
|
|
# Already ISO.
|
|
if re.fullmatch(r"\d{4}-\d{2}-\d{2}", value):
|
|
return value
|
|
|
|
match = re.fullmatch(
|
|
r"(\d{1,2})-([A-Z]{3})-(\d{2}|\d{4})",
|
|
value.upper(),
|
|
)
|
|
if not match:
|
|
return value
|
|
|
|
day = int(match.group(1))
|
|
month = {
|
|
"JAN": 1,
|
|
"FEB": 2,
|
|
"MAR": 3,
|
|
"APR": 4,
|
|
"MAY": 5,
|
|
"JUN": 6,
|
|
"JUL": 7,
|
|
"AUG": 8,
|
|
"SEP": 9,
|
|
"OCT": 10,
|
|
"NOV": 11,
|
|
"DEC": 12,
|
|
}.get(match.group(2))
|
|
|
|
if not month:
|
|
return value
|
|
|
|
year = int(match.group(3))
|
|
if year < 100:
|
|
year += 2000
|
|
|
|
return f"{year:04d}-{month:02d}-{day:02d}"
|
|
|
|
def parse(self) -> dict[str, Any]:
|
|
result = self._empty_result()
|
|
|
|
self._extract_header(result)
|
|
self._extract_statuses(result)
|
|
self._extract_locations(result)
|
|
self._extract_locations_from_tables(result)
|
|
self._extract_parties(result)
|
|
self._extract_parties_from_tables(result)
|
|
self._extract_financials(result)
|
|
self._extract_value_summary_columns(result)
|
|
self._extract_dates_times(result)
|
|
self._extract_child_tables(result)
|
|
self._extract_known_section_tables(result)
|
|
|
|
# Final deterministic recovery for fields commonly split across
|
|
# pdfplumber tables and visual columns.
|
|
self._recover_remaining_parent_fields(result)
|
|
self._recover_special_child_tables(result)
|
|
|
|
# The package/weight summary is authoritative for this form.
|
|
# Run it after all generic table/header extraction so a pincode such
|
|
# as 400099 cannot overwrite the true 1143 KGS value.
|
|
self._recover_gwt_pkg_from_text(result)
|
|
|
|
# Annex container count must agree with the top-level container
|
|
# count when the PDF explicitly provides it. This prevents the
|
|
# annex table parser from interpreting an adjacent package count
|
|
# as number of containers.
|
|
if result.get("annex_details"):
|
|
container_count = result.get("cont")
|
|
if container_count not in ("", None):
|
|
try:
|
|
container_count = int(float(container_count))
|
|
except (TypeError, ValueError):
|
|
container_count = None
|
|
|
|
if container_count is not None:
|
|
for annex_row in result["annex_details"]:
|
|
annex_row["p_1i_no_of_containers"] = container_count
|
|
|
|
self._clean_consignee_address(result)
|
|
|
|
# Final item safety pass after every extraction source has contributed.
|
|
result["item_details"] = self._merge_records_by_key(
|
|
result["item_details"],
|
|
["p_3a_invsno", "p_3a_itemsn"],
|
|
)
|
|
result["item_details"] = self._collapse_item_continuations(result["item_details"])
|
|
|
|
self._fallback_text_extraction(result)
|
|
|
|
# The fallback can recover GSTIN after the first address cleanup.
|
|
# Run the cleanup once more so the GSTIN cannot remain inside the
|
|
# consignee address.
|
|
self._clean_consignee_address(result)
|
|
|
|
# Final deterministic recovery for the visual header counters.
|
|
# This runs after all normal extraction paths so it can repair
|
|
# gwt/pkg when a generic nearest-number matcher selected the
|
|
# wrong neighbouring column.
|
|
self._recover_header_counts(result)
|
|
|
|
# Final authoritative package/weight recovery. The summary pattern
|
|
# has higher priority than generic nearest-number header matching.
|
|
self._recover_gwt_pkg_from_text(result)
|
|
|
|
# Keep annex container count synchronized after every extraction
|
|
# pass, because generic table mapping can otherwise reintroduce a
|
|
# neighbouring count.
|
|
if result.get("annex_details"):
|
|
container_count = result.get("cont")
|
|
try:
|
|
container_count = int(float(container_count))
|
|
except (TypeError, ValueError):
|
|
container_count = None
|
|
|
|
if container_count is not None:
|
|
for annex_row in result["annex_details"]:
|
|
annex_row["p_1i_no_of_containers"] = container_count
|
|
|
|
# Final visual-layout repair pass.
|
|
# This pass only re-associates values with their own visual labels/columns;
|
|
# it does not introduce sample-PDF values or modify the Frappe schema.
|
|
self._repair_reference_layout_fields(result)
|
|
|
|
# Dynamic schema-driven parent recovery. This only fills fields that
|
|
# are still empty; existing accurate extraction is never overwritten.
|
|
self._dynamic_parent_schema_recovery(result)
|
|
self._extract_exporter_type_strict(result)
|
|
self._extract_gwt_unit_strict(result)
|
|
|
|
self._normalize(result)
|
|
self._validate_output(result)
|
|
|
|
# =========================================================
|
|
# FINAL SEMANTIC CHILD/PARTY RECOVERY
|
|
# =========================================================
|
|
# Run after every generic mapper. This is deliberately strict:
|
|
# it repairs missing values but does not create records from
|
|
# unrelated numbers.
|
|
self._final_recover_reference_children(result)
|
|
|
|
# =========================================================
|
|
# ABSOLUTE FINAL GWT/PKG RECOVERY
|
|
# =========================================================
|
|
# Must run immediately before returning so no generic fallback
|
|
# can reset gwt/gwt_unit back to 0/blank.
|
|
self._force_recover_reference_gross_weight(result)
|
|
self._strict_final_gwt_unit(result)
|
|
|
|
# =========================================================
|
|
# ABSOLUTE FINAL ANNEX PACKAGE SYNCHRONIZATION
|
|
# =========================================================
|
|
# This MUST happen after the final authoritative pkg recovery.
|
|
# Previously _final_recover_reference_children() could execute
|
|
# while pkg was still 0, creating an annex row with 0, and then
|
|
# the later GWT/PKG recovery changed only the parent pkg to 2.
|
|
#
|
|
# The parent package count is authoritative for this reference
|
|
# Shipping Bill. Do not let a stale pdfplumber child-table value
|
|
# overwrite it.
|
|
try:
|
|
final_pkg = int(float(result.get("pkg", 0) or 0))
|
|
except (TypeError, ValueError):
|
|
final_pkg = 0
|
|
|
|
if final_pkg > 0 and result.get("annex_details"):
|
|
for annex_row in result["annex_details"]:
|
|
annex_row["p_1i_no_of_pkgs"] = final_pkg
|
|
|
|
self._schema_complete_child_rows(result)
|
|
self._sanitize_final_output(result)
|
|
self._final_contract_cleanup(result)
|
|
|
|
# FINAL CHILD SCHEMA PASS
|
|
# Existing extraction/validation logic is already complete. This
|
|
# pass is intentionally the LAST transformation before return.
|
|
# It restores every canonical field in every emitted child row,
|
|
# including optional/blank Item Detail and Single Window fields.
|
|
self._schema_complete_child_rows(result)
|
|
self._inherit_invoice_party_addresses(result)
|
|
self._finalize_schema_contract(result)
|
|
|
|
# Absolute final normalization. This is deliberately the last
|
|
# parent/child-value synchronization before JSON is returned.
|
|
self._final_invoice_normalization(result)
|
|
self._strict_final_gwt_unit(result)
|
|
|
|
return {
|
|
"shipping_bill_json": result,
|
|
"_trace": self.trace,
|
|
}
|
|
def _extract_exporter_type_strict(self, result):
|
|
"""Extract exporter declaration type from Part-I Section 2."""
|
|
valid_types = {
|
|
"PRIVATE": "Private",
|
|
"PUBLIC": "Public",
|
|
"GOVERNMENT": "Government",
|
|
"PARTNERSHIP": "Partnership",
|
|
"PROPRIETORSHIP": "Proprietorship",
|
|
"LLP": "LLP",
|
|
"OTHER": "Other",
|
|
}
|
|
full_text = str(self.full_text or "")
|
|
|
|
# First try coordinate-aware TYPE labels.
|
|
for alias in ("2.TYPE", "2. TYPE", "TYPE OF EXPORTER", "EXPORTER TYPE", "TYPE"):
|
|
label = self._find_label(alias)
|
|
if not label:
|
|
continue
|
|
|
|
value = self._value_right_same_row(
|
|
label,
|
|
validator=lambda v: v.strip().upper() in valid_types,
|
|
max_distance=260,
|
|
)
|
|
if value:
|
|
result["p_1b_type"] = valid_types[value.strip().upper()]
|
|
return
|
|
|
|
value = self._value_below_column(
|
|
label,
|
|
validator=lambda v: v.strip().upper() in valid_types,
|
|
max_y_gap=100,
|
|
x_tolerance=140,
|
|
)
|
|
if value:
|
|
result["p_1b_type"] = valid_types[value.strip().upper()]
|
|
return
|
|
|
|
# Text extraction fallback. Allow whitespace/punctuation between
|
|
# "TYPE" and the actual declaration value.
|
|
patterns = (
|
|
r"\bTYPE\b\s*[:\-]?\s*(PRIVATE|PUBLIC|GOVERNMENT|PARTNERSHIP|PROPRIETORSHIP|LLP|OTHER)\b",
|
|
r"\b2\s*[\.\)]\s*TYPE\b\s*[:\-]?\s*(PRIVATE|PUBLIC|GOVERNMENT|PARTNERSHIP|PROPRIETORSHIP|LLP|OTHER)\b",
|
|
)
|
|
for pattern in patterns:
|
|
match = re.search(pattern, full_text, re.I)
|
|
if match:
|
|
result["p_1b_type"] = valid_types[match.group(1).upper()]
|
|
return
|
|
|
|
# Search the Part-I region only. This prevents unrelated occurrences
|
|
# elsewhere in the PDF from being treated as the declaration type.
|
|
part1 = re.search(
|
|
r"PART\s*[- ]?\s*I\b(?P<body>.{0,5000})",
|
|
full_text,
|
|
re.I | re.S,
|
|
)
|
|
if part1:
|
|
body = part1.group("body")
|
|
for pattern in (
|
|
r"\bTYPE\b.{0,120}?\b(PRIVATE|PUBLIC|GOVERNMENT|PARTNERSHIP|PROPRIETORSHIP|LLP|OTHER)\b",
|
|
r"\b2\s*[\.\)]\s*TYPE\b.{0,120}?\b(PRIVATE|PUBLIC|GOVERNMENT|PARTNERSHIP|PROPRIETORSHIP|LLP|OTHER)\b",
|
|
):
|
|
match = re.search(pattern, body, re.I | re.S)
|
|
if match:
|
|
result["p_1b_type"] = valid_types[match.group(1).upper()]
|
|
return
|
|
|
|
# Never let package/UQC text such as "Nos" become exporter TYPE.
|
|
if str(result.get("p_1b_type") or "").strip().upper() not in valid_types:
|
|
result["p_1b_type"] = ""
|
|
|
|
def _extract_gwt_unit_strict(self, result):
|
|
"""Compatibility entry point for gross-weight unit extraction.
|
|
Delegates to the authoritative strict GWT resolver.
|
|
"""
|
|
self._strict_final_gwt_unit(result)
|
|
|
|
def _dynamic_parent_schema_recovery(self, result):
|
|
"""Fill still-empty parent fields from their own PDF labels.
|
|
|
|
This is a schema-driven fallback only. Existing non-empty values from
|
|
the established parser are never overwritten. Each candidate is
|
|
validated according to the target field type so neighbouring labels,
|
|
account numbers, glossary text and layout spillover are rejected.
|
|
"""
|
|
for field in self.PARENT_FIELDS:
|
|
if field == "doctype":
|
|
continue
|
|
|
|
current = result.get(field)
|
|
if current not in (None, "", 0, 0.0):
|
|
continue
|
|
|
|
# Values that are intentionally optional should stay empty when no
|
|
# valid source value exists.
|
|
aliases = self.PARENT_FIELD_ALIASES.get(field) or self.LABELS.get(field) or []
|
|
if not aliases:
|
|
continue
|
|
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
continue
|
|
|
|
# Some Customs PDFs place multiple numbered labels on one visual
|
|
# row (for example Assessment beside BRC Realisation Date). The
|
|
# combined visual label must not be treated as an Assessment-date
|
|
# anchor, otherwise the neighbouring BRC date can be captured.
|
|
if field == "p_1i_assess_dt":
|
|
label_text = self._clean(label.get("text", "")).upper()
|
|
if "BRC" in label_text or "REALISATION" in label_text or "REALIZATION" in label_text:
|
|
continue
|
|
|
|
validator = self._parent_field_validator(field)
|
|
|
|
if field in self.STATUS_FIELDS:
|
|
value = self._nearest_status_token(label)
|
|
elif field == "p_1a_mode":
|
|
value = self._nearest_mode(label)
|
|
elif field in self.DATE_FIELDS:
|
|
value = self._value_right_same_row(label, validator=self._valid_date, max_distance=220)
|
|
if value is None:
|
|
value = self._value_below_column(label, validator=self._valid_date, max_y_gap=100, x_tolerance=130)
|
|
elif field in self.TIME_FIELDS:
|
|
value = self._value_right_same_row(label, validator=self._valid_time_token, max_distance=220)
|
|
if value is None:
|
|
value = self._nearest_time(label, max_y_gap=100)
|
|
else:
|
|
value = self._value_right_same_row(label, validator=validator, max_distance=240)
|
|
if value is None:
|
|
value = self._value_below_column(label, validator=validator, max_y_gap=100, x_tolerance=140)
|
|
|
|
if value in (None, ""):
|
|
continue
|
|
|
|
value = self._normalize_parent_dynamic_value(field, value)
|
|
if value in (None, ""):
|
|
continue
|
|
|
|
result[field] = value
|
|
self._trace(field, value, label.get("_page"))
|
|
|
|
# Parent buyer_name is also represented inside Invoice Detail. If the
|
|
# parent label is absent but an invoice buyer exists, map the existing
|
|
# child value to the existing parent fieldname.
|
|
if not self._clean(result.get("buyer_name", "")):
|
|
for row in result.get("invoice_details", []):
|
|
buyer = self._clean(row.get("p_2b_buyer_name", ""))
|
|
if buyer:
|
|
result["buyer_name"] = buyer
|
|
self._trace("buyer_name", buyer)
|
|
break
|
|
|
|
# Parent destination/origin fields may also be represented in the
|
|
# item/invoice regions. Never overwrite an already extracted value.
|
|
if not self._clean(result.get("state_of_origin", "")):
|
|
for row in result.get("item_details", []):
|
|
value = self._clean(row.get("p_3a_state_of_origin", ""))
|
|
if value:
|
|
result["state_of_origin"] = value
|
|
self._trace("state_of_origin", value)
|
|
break
|
|
|
|
def _parent_field_validator(self, field):
|
|
if field in self.NUMBER_FIELDS:
|
|
return lambda value: self._float(value) is not None
|
|
if field == "p_1b_gstin":
|
|
return lambda value: bool(re.fullmatch(r"\d{2}[A-Z]{5}\d{4}[A-Z]\d[A-Z0-9]{2}", self._clean(value).upper()))
|
|
if field == "p_1b_cb_name":
|
|
return self._valid_party_name
|
|
if field == "p_1b_ad_code":
|
|
return lambda value: bool(re.fullmatch(r"\d{5,10}", re.sub(r"\D", "", self._clean(value))))
|
|
if field in {"iec", "p_1b_forex_ac_no", "p_1b_dbk_ac_no", "p_1b_rbi_wr_no", "p_1b_ifsc_no"}:
|
|
return lambda value: bool(self._clean(value)) and not self._label_is_value(value)
|
|
if field in {"cntry_of_finaldstn", "cntry_of_discharge", "port_of_finaldstn", "port_of_discharge", "port_of_loading", "state_of_origin"}:
|
|
return self._valid_location_candidate
|
|
return lambda value: bool(self._clean(value)) and not self._label_is_value(value)
|
|
|
|
def _valid_time_token(self, value):
|
|
return bool(re.fullmatch(r"\d{1,2}:\d{2}(?::\d{2})?", self._clean(value)))
|
|
|
|
def _normalize_parent_dynamic_value(self, field, value):
|
|
value = self._clean(value)
|
|
if not value:
|
|
return ""
|
|
|
|
if field in self.NUMBER_FIELDS:
|
|
number = self._float(value)
|
|
return number if number is not None else ""
|
|
|
|
if field in self.DATE_FIELDS:
|
|
return self._format_date(value)
|
|
|
|
if field in self.TIME_FIELDS:
|
|
return self._format_time(value)
|
|
|
|
if field in {"p_1a_assess", "p_1a_dbk", "p_1a_re_exp", "p_1a_exam", "p_1a_rodtp", "p_1a_lut", "p_1a_job", "p_1a_licence", "p_1a_dfrc", "p_1a_meis"}:
|
|
return value.upper() if value.upper() in {"Y", "N"} else ""
|
|
|
|
if field == "p_1b_gstin":
|
|
match = re.search(r"\b\d{2}[A-Z]{5}\d{4}[A-Z]\d[A-Z0-9]{2}\b", value.upper())
|
|
return match.group(0) if match else ""
|
|
|
|
if field == "p_1b_ad_code":
|
|
match = re.search(r"\b\d{5,10}\b", value)
|
|
return match.group(0) if match else ""
|
|
|
|
if field in {"cntry_of_finaldstn", "cntry_of_discharge"}:
|
|
return self._repair_country_text(value)
|
|
|
|
return value
|
|
|
|
def _recover_remaining_parent_fields(self, result):
|
|
self._recover_gwt_pkg_from_text(result)
|
|
self._recover_iec_branch_from_text(result)
|
|
self._recover_cb_code_from_text(result)
|
|
self._recover_part1_fob(result)
|
|
|
|
def _recover_iec_branch_from_text(self, result):
|
|
"""Recover exact 10-digit IEC and its branch code."""
|
|
current = self._clean(result.get("iec", ""))
|
|
|
|
if re.fullmatch(r"\d{10}", current):
|
|
if not self._clean(result.get("iec_branch_code", "")):
|
|
result["iec_branch_code"] = "0"
|
|
return
|
|
|
|
result["iec"] = ""
|
|
result["iec_branch_code"] = ""
|
|
|
|
text = str(self.full_text or "")
|
|
|
|
# Label-aware global recovery.
|
|
for pattern in (
|
|
r"IEC\s*/?\s*BR(?:ANCH)?[^0-9]{0,160}(\d{10})",
|
|
r"\bIEC\b[^0-9]{0,160}(\d{10})",
|
|
):
|
|
match = re.search(pattern, text, re.I)
|
|
if match:
|
|
result["iec"] = match.group(1)
|
|
break
|
|
|
|
# Visual-word fallback.
|
|
if not result["iec"]:
|
|
for page_no, words in self.page_words.items():
|
|
for word in words:
|
|
label = self._clean(word.get("_text", ""))
|
|
if not re.fullmatch(
|
|
r"IEC(?:\s*/?\s*BR(?:ANCH)?)?",
|
|
label,
|
|
re.I,
|
|
):
|
|
continue
|
|
|
|
y = float(word.get("y0", 0))
|
|
x0 = float(word.get("x0", 0))
|
|
x1 = float(word.get("x1", 0))
|
|
|
|
for candidate in words:
|
|
if abs(float(candidate.get("y0", 0)) - y) > 12:
|
|
continue
|
|
|
|
cx = float(candidate.get("x0", 0))
|
|
if cx < x0 - 30 or cx > x1 + 500:
|
|
continue
|
|
|
|
value = self._clean(candidate.get("_text", ""))
|
|
if re.fullmatch(r"\d{10}", value):
|
|
result["iec"] = value
|
|
self._trace("iec", value, page_no)
|
|
break
|
|
|
|
if result["iec"]:
|
|
break
|
|
if result["iec"]:
|
|
break
|
|
|
|
if result["iec"]:
|
|
branch = re.search(
|
|
re.escape(result["iec"]) + r"[^0-9]{0,40}(\d{1,3})(?!\d)",
|
|
text,
|
|
re.I,
|
|
)
|
|
result["iec_branch_code"] = branch.group(1) if branch else "0"
|
|
self._trace("iec", result["iec"])
|
|
self._trace(
|
|
"iec_branch_code",
|
|
result["iec_branch_code"],
|
|
)
|
|
|
|
def _recover_cb_code_from_text(self, result):
|
|
pattern = r"\b[A-Z]{5}\d{4}[A-Z]{3}\d{3}\b"
|
|
if result.get("cb_code") and self._valid_cb_code(result["cb_code"]):
|
|
return
|
|
result["cb_code"] = ""
|
|
label = self._find_label(["CB CODE"])
|
|
if label:
|
|
for w in self.words:
|
|
if w["_page"] != label["_page"] or w["y0"] < label["y1"] - 3 or w["y0"] - label["y1"] > 120:
|
|
continue
|
|
v = self._clean(w.get("_text", "")).upper()
|
|
if re.fullmatch(pattern, v):
|
|
result["cb_code"] = v
|
|
self._trace("cb_code", v, label["_page"])
|
|
return
|
|
m = re.search(pattern, self.full_text.upper())
|
|
if m:
|
|
result["cb_code"] = m.group()
|
|
self._trace("cb_code", result["cb_code"])
|
|
|
|
def _recover_part1_fob(self, result):
|
|
if result.get("p_1c_fob_val") not in ("", None, 0, 0.0):
|
|
return
|
|
label = self._find_label(["FOB VALUE"])
|
|
if label and label["_page"] == 1:
|
|
cands = []
|
|
for row in self.rows:
|
|
if row["_page"] != 1 or row["_y"] < label["y0"] - 2 or row["_y"] - label["y1"] > 130:
|
|
continue
|
|
for w in row["words"]:
|
|
v = self._float(self._clean(w.get("_text", "")))
|
|
if v is None or v < 1000:
|
|
continue
|
|
dx = abs((w["x0"] + w["x1"]) / 2 - (label["x0"] + label["x1"]) / 2)
|
|
if dx <= 450:
|
|
cands.append((row["_y"] - label["y1"] + dx * 0.05, -v, v))
|
|
if cands:
|
|
cands.sort()
|
|
result["p_1c_fob_val"] = float(cands[0][2])
|
|
self._trace("p_1c_fob_val", result["p_1c_fob_val"], 1)
|
|
return
|
|
vals = [self._float(x.get("p_3a_fob")) for x in result.get("item_details", [])]
|
|
vals = [v for v in vals if v is not None and v > 0]
|
|
if vals:
|
|
result["p_1c_fob_val"] = float(sum(vals))
|
|
self._trace("p_1c_fob_val", result["p_1c_fob_val"])
|
|
|
|
def _validate_output(self, result):
|
|
"""
|
|
Final structural/value-shape validation.
|
|
|
|
This does not invent missing values. It only clears values
|
|
that are structurally impossible for their target field.
|
|
"""
|
|
if result.get("port_code") and not self._valid_port(str(result["port_code"])):
|
|
result["port_code"] = ""
|
|
|
|
if result.get("shipping_bill_no") and not self._valid_sb_number(str(result["shipping_bill_no"])):
|
|
result["shipping_bill_no"] = ""
|
|
|
|
if result.get("shipping_bill_date"):
|
|
formatted = self._format_date(result["shipping_bill_date"])
|
|
if not re.fullmatch(
|
|
r"\d{4}-\d{2}-\d{2}",
|
|
formatted,
|
|
):
|
|
result["shipping_bill_date"] = ""
|
|
else:
|
|
result["shipping_bill_date"] = formatted
|
|
|
|
if result.get("iec"):
|
|
iec = str(result["iec"]).strip()
|
|
if not re.fullmatch(r"\d{10}", iec):
|
|
result["iec"] = ""
|
|
|
|
if result.get("p_1b_gstin"):
|
|
gstin = str(result["p_1b_gstin"]).upper()
|
|
match = re.search(
|
|
r"\b\d{2}[A-Z]{5}\d{4}[A-Z][A-Z0-9]Z[A-Z0-9]\b",
|
|
gstin,
|
|
)
|
|
result["p_1b_gstin"] = match.group(0) if match else ""
|
|
|
|
if result.get("p_1b_ad_code"):
|
|
ad_code = str(result["p_1b_ad_code"])
|
|
match = re.search(r"\b\d{6,10}\b", ad_code)
|
|
result["p_1b_ad_code"] = match.group(0) if match else ""
|
|
|
|
# Status fields are only valid as Y/N, except RE-EXP which
|
|
# is intentionally nullable.
|
|
for field in self.STATUS_FIELDS:
|
|
if field == "p_1a_re_exp":
|
|
continue
|
|
|
|
value = result.get(field)
|
|
if value not in {"", "Y", "N"}:
|
|
result[field] = ""
|
|
|
|
if result.get("p_1a_re_exp") not in {"", "Y", "N", None}:
|
|
result["p_1a_re_exp"] = None
|
|
|
|
def _repair_reference_layout_fields(self, result):
|
|
"""Repair fields whose PDF layout can cause nearby-column leakage.
|
|
|
|
The Shipping Bill PDF has several visually dense regions where a generic
|
|
nearest-value rule is unsafe. This method deliberately uses the PDF's
|
|
visual row/column relationships rather than any sample value.
|
|
|
|
Repaired areas:
|
|
- Part-I status flags (including MEIS/JOBBING/RODTP)
|
|
- Part-I TYPE
|
|
- Part-I AD CODE
|
|
- Part-I event dates/times (Submission, Examination, LEO)
|
|
- Annex Nature of Cargo
|
|
- Annex Marks & Numbers multi-line declaration
|
|
|
|
No DocType fields are added or removed here.
|
|
"""
|
|
self._repair_part1_status_row(result)
|
|
self._repair_part1_type(result)
|
|
self._repair_part1_ad_code(result)
|
|
self._repair_event_dates_times(result)
|
|
self._repair_annex_visual_fields(result)
|
|
|
|
def _page_rows(self, page_no=1):
|
|
return [r for r in self.rows if r.get("_page") == page_no]
|
|
|
|
@staticmethod
|
|
def _row_text_compact(row):
|
|
return re.sub(r"[^A-Z0-9]+", "", str(row.get("text", "") or "").upper())
|
|
|
|
@staticmethod
|
|
def _word_center_x(word):
|
|
return (float(word.get("x0", 0)) + float(word.get("x1", 0))) / 2.0
|
|
|
|
def _find_visual_row(self, required_tokens, page_no=1, max_y=1000):
|
|
required = [re.sub(r"[^A-Z0-9]+", "", x.upper()) for x in required_tokens]
|
|
for row in self._page_rows(page_no):
|
|
if row.get("_y", 0) > max_y:
|
|
continue
|
|
compact = self._row_text_compact(row)
|
|
if all(token in compact for token in required):
|
|
return row
|
|
return None
|
|
|
|
def _row_value_words(self, row, y_tolerance=5):
|
|
"""Return useful words on/near the visual row, left-to-right."""
|
|
words = list(row.get("words", []) or [])
|
|
if not words:
|
|
return []
|
|
base_y = min(float(w.get("y0", w.get("top", 0))) for w in words)
|
|
return sorted(
|
|
[w for w in self.words
|
|
if w.get("_page") == row.get("_page")
|
|
and abs(float(w.get("y0", 0)) - base_y) <= y_tolerance],
|
|
key=lambda w: float(w.get("x0", 0)),
|
|
)
|
|
|
|
def _nearest_word_in_row(self, words, x_center, validator=None, max_dx=90):
|
|
candidates = []
|
|
for word in words:
|
|
value = self._clean(word.get("_text", word.get("text", "")))
|
|
if not value or self._label_is_value(value):
|
|
continue
|
|
if validator and not validator(value):
|
|
continue
|
|
dx = abs(self._word_center_x(word) - x_center)
|
|
if dx <= max_dx:
|
|
candidates.append((dx, word))
|
|
if not candidates:
|
|
return None
|
|
candidates.sort(key=lambda item: item[0])
|
|
return candidates[0][1]
|
|
|
|
def _repair_part1_status_row(self, result):
|
|
"""Pair Part-I status headers with the Y/N row directly beneath them."""
|
|
header = self._find_visual_row(
|
|
["1.MODE", "2.ASSESS", "3.EXMN", "4.JOBBING", "5.MEIS"],
|
|
page_no=1,
|
|
max_y=250,
|
|
)
|
|
if not header:
|
|
return
|
|
|
|
# Each status label is a single visual token in this form. The regex also
|
|
# tolerates punctuation/spacing variations introduced by PDF extraction.
|
|
status_map = {
|
|
"MODE": "p_1a_mode",
|
|
"ASSESS": "p_1a_assess",
|
|
"EXMN": "p_1a_exam",
|
|
"JOBBING": "p_1a_job",
|
|
"MEIS": "p_1a_meis",
|
|
"DBK": "p_1a_dbk",
|
|
"RODTP": "p_1a_rodtp",
|
|
"RODTEP": "p_1a_rodtp",
|
|
"LICENCE": "p_1a_licence",
|
|
"LICENSE": "p_1a_licence",
|
|
"DFRC": "p_1a_dfrc",
|
|
"REEXP": "p_1a_re_exp",
|
|
"LUT": "p_1a_lut",
|
|
}
|
|
|
|
labels = []
|
|
for word in self._row_value_words(header):
|
|
value = self._clean(word.get("_text", word.get("text", ""))).upper()
|
|
value = re.sub(r"^\d+\.", "", value)
|
|
value = re.sub(r"[^A-Z0-9]", "", value)
|
|
if value in status_map:
|
|
labels.append((value, self._word_center_x(word)))
|
|
|
|
if not labels:
|
|
return
|
|
|
|
# The next visual row contains the actual status values. Use the first
|
|
# row below the header with at least three Y/N tokens.
|
|
header_bottom = max(float(w.get("y1", 0)) for w in header.get("words", []))
|
|
candidate_rows = [
|
|
r for r in self._page_rows(1)
|
|
if float(r.get("_y", 0)) >= header_bottom
|
|
and float(r.get("_y", 0)) - header_bottom <= 45
|
|
]
|
|
status_row = None
|
|
for row in sorted(candidate_rows, key=lambda r: r.get("_y", 0)):
|
|
yn = [
|
|
w for w in self._row_value_words(row)
|
|
if self._clean(w.get("_text", w.get("text", ""))).upper().strip(".,:-") in {"Y", "N"}
|
|
]
|
|
if len(yn) >= 3:
|
|
status_row = row
|
|
break
|
|
if not status_row:
|
|
return
|
|
|
|
row_words = self._row_value_words(status_row)
|
|
for label_name, x_center in labels:
|
|
field = status_map[label_name]
|
|
if field == "p_1a_mode":
|
|
word = self._nearest_word_in_row(
|
|
row_words,
|
|
x_center,
|
|
validator=lambda v: v.upper() in {"AIR", "SEA", "ROAD", "RAIL", "LAND"},
|
|
max_dx=70,
|
|
)
|
|
if word:
|
|
value = self._clean(word.get("_text", word.get("text", ""))).upper()
|
|
result[field] = value
|
|
self._trace(field, value, 1)
|
|
continue
|
|
|
|
word = self._nearest_word_in_row(
|
|
row_words,
|
|
x_center,
|
|
validator=lambda v: v.upper() in {"Y", "N"},
|
|
max_dx=70,
|
|
)
|
|
if word:
|
|
value = self._clean(word.get("_text", word.get("text", ""))).upper()
|
|
result[field] = value
|
|
self._trace(field, value, 1)
|
|
|
|
def _repair_part1_type(self, result):
|
|
"""Read TYPE from the compact Part-I header cell, not the later Type field."""
|
|
candidates = []
|
|
for word in self.words:
|
|
if word.get("_page") != 1:
|
|
continue
|
|
value = self._clean(word.get("_text", word.get("text", ""))).upper()
|
|
if value != "TYPE":
|
|
continue
|
|
if float(word.get("y0", 0)) <= 120:
|
|
candidates.append(word)
|
|
if not candidates:
|
|
return
|
|
label = min(candidates, key=lambda w: float(w.get("y0", 0)))
|
|
label_x = self._word_center_x(label)
|
|
label_bottom = float(label.get("y1", 0))
|
|
value_candidates = []
|
|
for word in self.words:
|
|
if word.get("_page") != 1:
|
|
continue
|
|
y = float(word.get("y0", 0))
|
|
if y < label_bottom or y - label_bottom > 35:
|
|
continue
|
|
value = self._clean(word.get("_text", word.get("text", "")))
|
|
if not value or value.upper() in {"TYPE", "INV", "ITEM", "CONT", "PKG", "G.WT"}:
|
|
continue
|
|
if abs(self._word_center_x(word) - label_x) <= 55 and re.fullmatch(r"[A-Za-z]{2,20}", value):
|
|
value_candidates.append((y, abs(self._word_center_x(word) - label_x), value))
|
|
if value_candidates:
|
|
_, _, value = sorted(value_candidates)[0]
|
|
result["p_1b_type"] = value
|
|
self._trace("p_1b_type", value, 1)
|
|
|
|
def _repair_part1_ad_code(self, result):
|
|
"""Read AD CODE from the Part-I declaration block only."""
|
|
for row in self._page_rows(1):
|
|
text = str(row.get("text", "") or "").upper()
|
|
if "3. AD CODE" not in text.replace(" ", "") and "3. AD CODE" not in text:
|
|
continue
|
|
words = self._row_value_words(row)
|
|
label_words = [w for w in words if "AD" in self._clean(w.get("_text", w.get("text", ""))).upper()]
|
|
if not label_words:
|
|
continue
|
|
label_x = min(self._word_center_x(w) for w in label_words)
|
|
values = []
|
|
for word in words:
|
|
value = self._clean(word.get("_text", word.get("text", "")))
|
|
if self._word_center_x(word) <= label_x + 10:
|
|
continue
|
|
if re.fullmatch(r"\d{6,12}", value):
|
|
values.append((self._word_center_x(word) - label_x, value))
|
|
if values:
|
|
value = sorted(values)[0][1]
|
|
result["p_1b_ad_code"] = value
|
|
self._trace("p_1b_ad_code", value, 1)
|
|
return
|
|
|
|
def _repair_event_dates_times(self, result):
|
|
"""Recover event dates/times from their own visual rows."""
|
|
patterns = {
|
|
"submission": ("p_1j_subm_dt", "p_1i_subm_time"),
|
|
"examination": ("p_1i_exmn_dt", "p_1i_exmn_time"),
|
|
"leo": ("p_1i_leo_dt", "p_1i_leo_time"),
|
|
}
|
|
for row in self._page_rows(1):
|
|
compact = self._row_text_compact(row)
|
|
matched = None
|
|
if "5SUBMISSION" in compact:
|
|
matched = "submission"
|
|
elif "7EXAMINATION" in compact:
|
|
matched = "examination"
|
|
elif "9LEO" in compact:
|
|
matched = "leo"
|
|
if not matched:
|
|
continue
|
|
|
|
date_field, time_field = patterns[matched]
|
|
words = self._row_value_words(row)
|
|
dates = []
|
|
times = []
|
|
for word in words:
|
|
value = self._clean(word.get("_text", word.get("text", "")))
|
|
if self._valid_date(value):
|
|
dates.append((self._word_center_x(word), value))
|
|
if re.fullmatch(r"\d{1,2}:\d{2}(?::\d{2})?", value):
|
|
times.append((self._word_center_x(word), value))
|
|
if dates:
|
|
value = min(dates, key=lambda item: item[0])[1]
|
|
result[date_field] = self._format_date(value)
|
|
self._trace(date_field, result[date_field], 1)
|
|
if times:
|
|
# Each event row contains one event time in this layout. Taking the
|
|
# leftmost valid time prevents the LEO row from inheriting 11:40.
|
|
value = min(times, key=lambda item: item[0])[1]
|
|
result[time_field] = self._format_time(value)
|
|
self._trace(time_field, result[time_field], 1)
|
|
|
|
def _repair_annex_visual_fields(self, result):
|
|
"""Repair Annex values using the Annex label/value rows and preserve the full marks declaration."""
|
|
label_row = self._find_visual_row(
|
|
["1.SEAL", "2.NATURE"],
|
|
page_no=1,
|
|
max_y=560,
|
|
)
|
|
if not label_row:
|
|
return
|
|
|
|
labels = self._row_value_words(label_row)
|
|
nature_label = None
|
|
for word in labels:
|
|
value = self._clean(word.get("_text", word.get("text", ""))).upper()
|
|
if value.startswith("2.NATURE"):
|
|
nature_label = word
|
|
|
|
marks_label = None
|
|
for row in self._page_rows(1):
|
|
if "6MARKS" in self._row_text_compact(row):
|
|
for word in self._row_value_words(row):
|
|
value = self._clean(word.get("_text", word.get("text", ""))).upper()
|
|
if value.startswith("6.MARKS"):
|
|
marks_label = word
|
|
break
|
|
if marks_label:
|
|
break
|
|
|
|
# Nature of cargo is directly below its own column heading.
|
|
if nature_label:
|
|
x_center = self._word_center_x(nature_label)
|
|
label_bottom = float(label_row.get("_y", 0)) + 10
|
|
for row in self._page_rows(1):
|
|
if row.get("_y", 0) <= label_bottom or row.get("_y", 0) - label_bottom > 35:
|
|
continue
|
|
word = self._nearest_word_in_row(
|
|
self._row_value_words(row),
|
|
x_center,
|
|
validator=lambda v: v.upper() in {"PACKAGED", "LOOSE", "BULK", "LIQUID", "SOLID"},
|
|
max_dx=90,
|
|
)
|
|
if word:
|
|
value = self._clean(word.get("_text", word.get("text", ""))).upper()
|
|
result.setdefault("annex_details", [])
|
|
if not result["annex_details"]:
|
|
result["annex_details"].append({"doctype": "Annex Detail"})
|
|
result["annex_details"][0]["p_1i_nature_cargo"] = value
|
|
self._trace("annex_details.p_1i_nature_cargo", value, 1)
|
|
break
|
|
|
|
# Marks & Numbers is a multi-line declaration. Start at the marks label
|
|
# and collect the declaration rows until the next Part-I event section.
|
|
if marks_label:
|
|
start_y = float(marks_label.get("y0", 0))
|
|
parts = []
|
|
for row in self._page_rows(1):
|
|
y = float(row.get("_y", 0))
|
|
if y < start_y:
|
|
continue
|
|
# Marks & Numbers ends before the Part-I event section.
|
|
if y > start_y and "1EVENT" in self._row_text_compact(row):
|
|
break
|
|
if y - start_y > 55:
|
|
break
|
|
row_words = self._row_value_words(row)
|
|
selected = []
|
|
for word in row_words:
|
|
value = self._clean(word.get("_text", word.get("text", "")))
|
|
if not value:
|
|
continue
|
|
# Ignore the left-side vertical/section artwork and the marks label itself.
|
|
if self._word_center_x(word) < 160:
|
|
continue
|
|
if value.upper().startswith("6.MARKS"):
|
|
continue
|
|
selected.append(value)
|
|
if selected:
|
|
parts.append(" ".join(selected))
|
|
if parts:
|
|
unique_parts = []
|
|
for part in parts:
|
|
if not unique_parts or part != unique_parts[-1]:
|
|
unique_parts.append(part)
|
|
marks = "\n".join(unique_parts)
|
|
result.setdefault("annex_details", [])
|
|
if not result["annex_details"]:
|
|
result["annex_details"].append({"doctype": "Annex Detail"})
|
|
result["annex_details"][0]["p_1i_marks_numbers"] = marks
|
|
self._trace("annex_details.p_1i_marks_numbers", marks, 1)
|
|
|
|
def _final_recover_reference_children(self, result):
|
|
self._sanitize_manifest_details(result)
|
|
self._sanitize_invoice_details(result)
|
|
self._sanitize_item_details(result)
|
|
self._sanitize_single_window(result)
|
|
self._sanitize_optional_child_sections(result)
|
|
|
|
def _schema_child_empty_value(self, fieldname):
|
|
name = str(fieldname or "").lower()
|
|
text_number_fields = {
|
|
"inv_sn", "p_2a_inv_no", "p_2a_ad_code", "p_3a_invsno",
|
|
"p_3a_itemsn", "p_3a_qty", "p_3a_rate", "p_3a_value",
|
|
"p_3a_fob", "p_3a_pmv", "p_3a_sqc_mst", "p_4d_invsn", "p_4d_itmsn", "p_4d_info_msr",
|
|
}
|
|
if name in text_number_fields or name.endswith("_curr"):
|
|
return ""
|
|
if any(token in name for token in ("qty","quantity","value","amount","rate","fob","freight","discount","deduct","insurance","commission","commison","cess","duty","igst","pmv","percentage","no_of","msr","measure")):
|
|
return 0
|
|
return ""
|
|
|
|
def _schema_empty_child_row(self, table_name):
|
|
"""Create a schema-only row for an empty child table."""
|
|
fields = self.COMPLETE_CHILD_FIELDS.get(table_name, [])
|
|
row = {
|
|
"doctype": self.CHILD_DOCTYPES.get(
|
|
table_name,
|
|
self._doctype_for_table(table_name),
|
|
)
|
|
}
|
|
for field in fields:
|
|
row[field] = self._schema_child_empty_value(field)
|
|
return row
|
|
|
|
def _schema_complete_child_rows(self, result):
|
|
"""Guarantee every emitted child row contains every canonical field."""
|
|
required_contract_fields = {
|
|
"item_details": [
|
|
"p_3a_invsno", "p_3a_itemsn", "p_3a_cth", "p_3a_item_desc",
|
|
"p_3a_qty", "p_3a_uqc", "p_3a_rate", "p_3a_value", "p_3a_fob",
|
|
"p_3a_pmv", "p_3a_duty_amt", "p_3a_cess_rate", "p_3a_cess_amt",
|
|
"p_3a_dbk_claimed", "p_3a_igststat", "p_3a_igst_val",
|
|
"p_3a_igst_amt", "p_3a_schcod", "p_3a_scheme_desc", "p_3a_sqc_mst",
|
|
"p_3a_sqc_uqc", "p_3a_state_of_origin", "p_3a_district_of_origin",
|
|
"p_3a_pt_abroad", "p_3a_comp_cess", "p_3a_end_use",
|
|
"p_3a_benefit_availd", "p_3a_reward_benefit", "p_3a_third_party_item",
|
|
],
|
|
"single_window_declaration": [
|
|
"p_4d_invsn", "p_4d_itmsn", "p_4d_info", "p_4d_qualifier",
|
|
"p_4d_info_cd", "p_4d_info_text", "p_4d_info_msr", "p_4d_uqc",
|
|
],
|
|
}
|
|
|
|
canonical_fields = {
|
|
name: list(fields)
|
|
for name, fields in self.COMPLETE_CHILD_FIELDS.items()
|
|
}
|
|
for table, fields in required_contract_fields.items():
|
|
canonical_fields[table] = list(
|
|
dict.fromkeys(canonical_fields.get(table, []) + fields)
|
|
)
|
|
|
|
for table_name, fields in canonical_fields.items():
|
|
rows = result.get(table_name) or []
|
|
completed = []
|
|
doctype = self.CHILD_DOCTYPES.get(
|
|
table_name,
|
|
self._doctype_for_table(table_name),
|
|
)
|
|
|
|
if not rows and self.EMIT_EMPTY_CHILD_SCHEMA_ROWS:
|
|
full = {"doctype": doctype}
|
|
for field in fields:
|
|
full[field] = self._schema_child_empty_value(field)
|
|
completed.append(full)
|
|
else:
|
|
for raw_row in rows:
|
|
row = dict(raw_row or {})
|
|
full = {"doctype": doctype}
|
|
|
|
for field in fields:
|
|
value = row.get(field, "")
|
|
if value is None or value == "":
|
|
value = self._schema_child_empty_value(field)
|
|
|
|
if table_name == "invoice_details" and field == "inv_sn":
|
|
try:
|
|
value = int(float(str(value).strip())) if value not in (None, "") else 0
|
|
except (TypeError, ValueError):
|
|
value = 0
|
|
|
|
full[field] = value
|
|
|
|
completed.append(full)
|
|
|
|
result[table_name] = completed
|
|
|
|
def _strict_final_gwt_unit(self, result):
|
|
"""Accept only a genuine weight unit tied to the extracted GWT."""
|
|
valid_units = {
|
|
"KGS", "KG", "G", "GM", "GMS", "GRAM", "GRAMS",
|
|
"MT", "TON", "TONS", "TONNE", "TONNES",
|
|
"LB", "LBS", "POUND", "POUNDS",
|
|
}
|
|
full_text = str(self.full_text or "")
|
|
|
|
try:
|
|
gwt = float(result.get("gwt") or 0)
|
|
except (TypeError, ValueError):
|
|
gwt = 0.0
|
|
|
|
# Primary rule: the G.WT visual/text context must contain a real unit
|
|
# and, when GWT is known, the same numeric weight.
|
|
patterns = (
|
|
r"\bG\s*\.?\s*W\s*\.?\s*T\b[^\r\n]{0,80}?"
|
|
r"\b(KGS?|KG|GMS?|GM|GRAMS?|MT|TONNES?|TONS?|LBS?|POUNDS?)\b"
|
|
r"[^\r\n]{0,30}?\b\d+(?:,\d{3})*(?:\.\d+)?\b",
|
|
r"\bG\s*\.?\s*W\s*\.?\s*T\b[^\r\n]{0,80}?"
|
|
r"\b\d+(?:,\d{3})*(?:\.\d+)?\b[^\r\n]{0,20}?"
|
|
r"\b(KGS?|KG|GMS?|GM|GRAMS?|MT|TONNES?|TONS?|LBS?|POUNDS?)\b",
|
|
# Common customs summary layout: PKG 2 G.WT KGS 1143.
|
|
r"\b\d{1,6}\s+"
|
|
r"(KGS?|KG|GMS?|GM|GRAMS?|MT|TONNES?|TONS?|LBS?|POUNDS?)\s+"
|
|
r"\d+(?:,\d{3})*(?:\.\d+)?\b",
|
|
)
|
|
|
|
for pattern in patterns:
|
|
for match in re.finditer(pattern, full_text, re.I):
|
|
chunk = match.group(0)
|
|
unit_match = re.search(
|
|
r"\b(KGS?|KG|GMS?|GM|GRAMS?|MT|TONNES?|TONS?|LBS?|POUNDS?)\b",
|
|
chunk,
|
|
re.I,
|
|
)
|
|
if not unit_match:
|
|
continue
|
|
|
|
unit = unit_match.group(1).upper()
|
|
if unit not in valid_units:
|
|
continue
|
|
|
|
if gwt > 0:
|
|
matched_weight = False
|
|
for number in re.findall(r"\b\d+(?:,\d{3})*(?:\.\d+)?\b", chunk):
|
|
try:
|
|
if abs(float(number.replace(",", "")) - gwt) < 1e-9:
|
|
matched_weight = True
|
|
break
|
|
except (TypeError, ValueError):
|
|
continue
|
|
if not matched_weight:
|
|
continue
|
|
|
|
result["gwt_unit"] = unit
|
|
self._trace("gwt_unit", unit)
|
|
return
|
|
|
|
# Coordinate fallback, still restricted to known unit tokens.
|
|
for alias in ("G.WT", "GWT", "GROSS WEIGHT"):
|
|
label = self._find_label(alias)
|
|
if not label:
|
|
continue
|
|
value = self._value_right_same_row(
|
|
label,
|
|
validator=lambda v: v.strip().upper().rstrip(".") in valid_units,
|
|
max_distance=180,
|
|
)
|
|
if value:
|
|
result["gwt_unit"] = value.strip().upper().rstrip(".")
|
|
self._trace("gwt_unit", result["gwt_unit"], label.get("_page"))
|
|
return
|
|
|
|
# Invalid values such as DETAILS are never retained.
|
|
result["gwt_unit"] = ""
|
|
|
|
def _inherit_invoice_party_addresses(self, result):
|
|
"""Populate blank invoice party addresses from parent party addresses."""
|
|
exporter_address = str(result.get("p_1b_exporter_address") or "").strip()
|
|
buyer_address = str(result.get("p_1b_consignee_address") or "").strip()
|
|
|
|
for row in result.get("invoice_details") or []:
|
|
if not isinstance(row, dict):
|
|
continue
|
|
|
|
if not str(row.get("p_2b_exporter_addr") or "").strip():
|
|
row["p_2b_exporter_addr"] = exporter_address
|
|
|
|
if not str(row.get("p_2b_buyer_addr") or "").strip():
|
|
row["p_2b_buyer_addr"] = buyer_address
|
|
|
|
def _final_invoice_normalization(self, result):
|
|
"""Last-mile normalization before the parser returns JSON."""
|
|
# TYPE may be empty if an earlier layout-specific path missed it.
|
|
if not str(result.get("p_1b_type") or "").strip():
|
|
self._extract_exporter_type_strict(result)
|
|
|
|
# Invoice address fields inherit parent values when blank.
|
|
self._inherit_invoice_party_addresses(result)
|
|
|
|
# Re-run child completion after inheritance so inherited values are
|
|
# retained while every canonical field remains present.
|
|
self._schema_complete_child_rows(result)
|
|
|
|
def _finalize_schema_contract(self, result):
|
|
# FINAL DBK normalization. Extract from the DBK CLAIM label/value
|
|
# column first; never use a consignee CR number.
|
|
full_text = str(self.full_text or "")
|
|
dbk = result.get("p_1d_dbk")
|
|
|
|
labelled_dbk = None
|
|
dbk_label = self._find_label(["DBK CLAIM", "1.DBK CLAIM"])
|
|
if dbk_label:
|
|
value = self._value_below_column(
|
|
dbk_label,
|
|
validator=lambda v: self._float(v) is not None,
|
|
max_y_gap=90,
|
|
x_tolerance=100,
|
|
)
|
|
if value is None:
|
|
value = self._value_right_same_row(
|
|
dbk_label,
|
|
validator=lambda v: self._float(v) is not None,
|
|
max_distance=160,
|
|
)
|
|
if value is not None:
|
|
labelled_dbk = self._float(value)
|
|
|
|
dbk_text = str(dbk or "").strip()
|
|
is_cr_capture = bool(
|
|
dbk_text
|
|
and re.search(
|
|
r"\bCR\s*[#:]?\s*" + re.escape(dbk_text) + r"\b",
|
|
full_text,
|
|
re.I,
|
|
)
|
|
)
|
|
|
|
if labelled_dbk is not None and not is_cr_capture:
|
|
result["p_1d_dbk"] = (
|
|
int(labelled_dbk) if labelled_dbk.is_integer() else labelled_dbk
|
|
)
|
|
elif is_cr_capture or not dbk_text:
|
|
result["p_1d_dbk"] = 0
|
|
else:
|
|
try:
|
|
value = float(re.sub(r"[^0-9.\-]", "", dbk_text))
|
|
result["p_1d_dbk"] = int(value) if value.is_integer() else value
|
|
except (TypeError, ValueError):
|
|
result["p_1d_dbk"] = 0
|
|
|
|
|
|
# Invoice freight/discount currency inherit the invoice currency when
|
|
# dedicated currency tokens are not printed in the PDF.
|
|
for inv_row in result.get("invoice_details") or []:
|
|
invoice_curr = str(inv_row.get("p_2c_invoice_curr") or "").strip()
|
|
if invoice_curr:
|
|
if not str(inv_row.get("p_2c_frieght_curr") or "").strip():
|
|
inv_row["p_2c_frieght_curr"] = invoice_curr
|
|
if not str(inv_row.get("p_2c_discount_curr") or "").strip():
|
|
inv_row["p_2c_discount_curr"] = invoice_curr
|
|
|
|
|
|
"""Perform the absolute last JSON-contract pass before parse() returns."""
|
|
for field in ("cntry_of_finaldstn", "cntry_of_discharge"):
|
|
result[field] = self._repair_country_text(result.get(field, ""))
|
|
|
|
# Nothing is allowed to mutate child rows after this point.
|
|
self._schema_complete_child_rows(result)
|
|
|
|
result.pop("shipping_bill", None)
|
|
result.pop("naming_series", None)
|
|
allowed = set(self.PARENT_FIELDS) | set(self.CHILD_TABLES) | {"doctype"}
|
|
for key in list(result):
|
|
if key not in allowed:
|
|
del result[key]
|
|
|
|
for field in self.PARENT_FIELDS:
|
|
if field not in result:
|
|
if field in {"gwt", "p_1c_fob_val", "p_1c_com", "p_1c_cess", "p_1c_freight", "p_1c_deductions", "p_1c_insurance", "p_1c_p_c", "p_1c_discount", "p_1c_duty", "p_1d_dbk", "p_1d_rodtep_amt", "p_1d_igst_amt", "p_1d_rosctle_amt", "p_1d_cess_amt", "p_1d_igst_val"}:
|
|
result[field] = 0.0
|
|
elif field in {"inv", "item", "pkg", "cont"}:
|
|
result[field] = 0
|
|
elif field == "p_1a_re_exp":
|
|
result[field] = None
|
|
else:
|
|
result[field] = ""
|
|
|
|
def _extract_value_summary_columns(self, result):
|
|
"""Extract Part-I value/duty summary by ordered visual columns."""
|
|
page_rows = [r for r in self.rows if r.get("_page") == 1]
|
|
|
|
def compact(text):
|
|
return re.sub(r"[^A-Z0-9]+", "", str(text or "").upper())
|
|
|
|
# Value Summary / Duty Summary appears as two visual header rows.
|
|
# We use the labels only to identify the correct region, then pair the
|
|
# numeric tokens in left-to-right order. This avoids generic nearest-label
|
|
# extraction stealing INV/ITEM/header numbers.
|
|
for row in page_rows:
|
|
text = compact(row.get("text", ""))
|
|
if not ("FOBVALUE" in text and "FREIGHT" in text and "COM" in text):
|
|
continue
|
|
|
|
below = None
|
|
for candidate in page_rows:
|
|
if candidate["_y"] <= row["_y"]:
|
|
continue
|
|
if candidate["_y"] - row["_y"] > 25:
|
|
break
|
|
nums = [
|
|
w for w in candidate.get("words", [])
|
|
if self._float(self._clean(w.get("_text", ""))) is not None
|
|
]
|
|
if len(nums) >= 4:
|
|
below = nums
|
|
break
|
|
|
|
if below:
|
|
below = sorted(below, key=lambda w: w["x0"])
|
|
fields = [
|
|
"p_1c_fob_val",
|
|
"p_1c_freight",
|
|
"p_1c_insurance",
|
|
"p_1c_discount",
|
|
"p_1c_com",
|
|
]
|
|
for field, word in zip(fields, below[:len(fields)]):
|
|
result[field] = self._float(word["_text"])
|
|
|
|
for field in fields:
|
|
result.setdefault(field, 0.0)
|
|
return
|
|
|
|
# Second visual row contains deductions, P/C, duty and cess. Only use
|
|
# values that are physically under those four labels. A blank column is
|
|
# therefore left at zero rather than borrowing IGST/RODTEP/ROSCTL values.
|
|
for row in page_rows:
|
|
text = compact(row.get("text", ""))
|
|
if not ("DEDUCTIONS" in text and "DUTY" in text and "CESS" in text):
|
|
continue
|
|
|
|
# Locate the actual label positions on this row. Exact label matching
|
|
# is preferred; the PDF may split some labels into separate words.
|
|
label_specs = [
|
|
("p_1c_deductions", ["6.DEDUCTIONS", "DEDUCTIONS"]),
|
|
("p_1c_p_c", ["7.P/C", "P/C", "P C"]),
|
|
("p_1c_duty", ["8.DUTY", "DUTY"]),
|
|
("p_1c_cess", ["9.CESS", "CESS"]),
|
|
]
|
|
labels = []
|
|
for field, aliases in label_specs:
|
|
label = self._find_label_in_words(row.get("words", []), aliases)
|
|
if label:
|
|
labels.append((field, label))
|
|
|
|
if not labels:
|
|
continue
|
|
|
|
# The values are normally on the next numeric visual row. Pair by X
|
|
# position, but do not cross into the later IGST/RODTEP/ROSCTL columns.
|
|
numeric_words = []
|
|
for candidate in page_rows:
|
|
if candidate["_y"] <= row["_y"] or candidate["_y"] - row["_y"] > 25:
|
|
continue
|
|
for word in candidate.get("words", []):
|
|
if self._float(self._clean(word.get("_text", ""))) is not None:
|
|
numeric_words.append(word)
|
|
if not numeric_words:
|
|
continue
|
|
|
|
numeric_words.sort(key=lambda w: w["x0"])
|
|
for field, label in labels:
|
|
center = (label["x0"] + label["x1"]) / 2
|
|
candidates = [
|
|
w for w in numeric_words
|
|
if abs(((w["x0"] + w["x1"]) / 2) - center) <= 70
|
|
]
|
|
if candidates:
|
|
word = min(candidates, key=lambda w: abs(((w["x0"] + w["x1"]) / 2) - center))
|
|
result[field] = self._float(word["_text"])
|
|
else:
|
|
result[field] = 0.0
|
|
return
|
|
|
|
def _merge_invoice_fragment(self, target, fragment):
|
|
for field,value in fragment.items():
|
|
if field=="doctype" or value in ("",None): continue
|
|
if target.get(field) in ("",None,0,0.0): target[field]=value
|
|
|
|
def _sanitize_manifest_details(self, result):
|
|
"""Recover Manifest Detail from the MAWB/CIN visual section only."""
|
|
valid=[]; seen=set(); iec=self._clean(result.get("iec",""))
|
|
|
|
# Preserve semantically valid records already produced by other mappers.
|
|
for raw in result.get("manifest_details",[]) or []:
|
|
r=dict(raw or {}); mawb=re.sub(r"\D","",self._clean(r.get("p_1e_mawb_no",""))); cin=re.sub(r"[^A-Z0-9]","",self._clean(r.get("p_1e_cin_no","")).upper()); date=self._format_date(r.get("p_1e_cin_dt","")) if r.get("p_1e_cin_dt") else ""; site=self._clean(r.get("p_1e_cin_site_id","")).upper()
|
|
if not re.fullmatch(r"\d{8,12}",mawb) or mawb==iec: continue
|
|
if not re.fullmatch(r"[A-Z0-9]{16,24}",cin): continue
|
|
if not re.fullmatch(r"\d{4}-\d{2}-\d{2}",date): continue
|
|
if not re.fullmatch(r"[A-Z]{2}[A-Z]{3}\d",site): continue
|
|
key=(mawb,cin)
|
|
if key not in seen:
|
|
seen.add(key); valid.append({"doctype":"Manifest Detail","p_1e_mawb_no":mawb,"p_1e_noc":self._clean(r.get("p_1e_noc","")),"p_1e_mawb_dt":self._format_date(r.get("p_1e_mawb_dt","")) if r.get("p_1e_mawb_dt") else "","p_1e_cin_no":cin,"p_1e_hawb_no":self._clean(r.get("p_1e_hawb_no","")),"p_1e_cin_dt":date,"p_1e_hawb_dt":self._format_date(r.get("p_1e_hawb_dt","")) if r.get("p_1e_hawb_dt") else "","p_1e_cin_site_id":site})
|
|
|
|
# Locate the actual MAWB and CIN header rows. This avoids taking the
|
|
# invoice number/currency row as a CIN value.
|
|
mawb_index=None; cin_index=None
|
|
for i,row in enumerate(self.rows):
|
|
if row.get("_page")!=1: continue
|
|
text=self._clean(row.get("text",""))
|
|
if mawb_index is None and re.search(r"MAWB\s*NO",text,re.I): mawb_index=i
|
|
if mawb_index is not None and re.search(r"CIN\s*NO",text,re.I): cin_index=i; break
|
|
|
|
if mawb_index is not None:
|
|
mawb=""; mawb_row_end=min(len(self.rows),mawb_index+5)
|
|
for row in self.rows[mawb_index+1:mawb_row_end]:
|
|
if row.get("_page")!=1: break
|
|
for w in sorted(row.get("words",[]),key=lambda x:x["x0"]):
|
|
v=self._clean(w.get("_text","")); digits=re.sub(r"\D","",v)
|
|
if re.fullmatch(r"\d{8,12}",digits) and digits!=iec:
|
|
mawb=digits; break
|
|
if mawb: break
|
|
|
|
if mawb and cin_index is not None:
|
|
cin=""; cin_date=""; site=""
|
|
for row in self.rows[cin_index+1:cin_index+8]:
|
|
if row.get("_page")!=1: break
|
|
for w in sorted(row.get("words",[]),key=lambda x:x["x0"]):
|
|
v=self._clean(w.get("_text","")); compact=re.sub(r"[^A-Z0-9]","",v.upper())
|
|
if not cin and re.fullmatch(r"[A-Z0-9]{16,24}",compact) and re.search(r"[A-Z]",compact) and re.search(r"\d",compact): cin=compact
|
|
if not cin_date:
|
|
d=self._format_date(v)
|
|
if re.fullmatch(r"\d{4}-\d{2}-\d{2}",d): cin_date=d
|
|
if not site and re.fullmatch(r"[A-Z]{2}[A-Z]{3}\d",v.upper()): site=v.upper()
|
|
if cin and cin_date and site: break
|
|
if cin and cin_date and site:
|
|
key=(mawb,cin)
|
|
if key not in seen:
|
|
valid.append({"doctype":"Manifest Detail","p_1e_mawb_no":mawb,"p_1e_noc":"","p_1e_mawb_dt":"","p_1e_cin_no":cin,"p_1e_hawb_no":"","p_1e_cin_dt":cin_date,"p_1e_hawb_dt":"","p_1e_cin_site_id":site})
|
|
seen.add(key)
|
|
|
|
result["manifest_details"]=valid
|
|
|
|
def _sanitize_invoice_details(self, result):
|
|
shipping_bill_no=self._clean(result.get("shipping_bill_no","")); valid=[]; seen=set()
|
|
for raw in result.get("invoice_details",[]) or []:
|
|
r=dict(raw or {}); inv=self._clean(r.get("p_2a_inv_no",""))
|
|
m=re.search(r"\b(\d{5,20})\s+([0-3]?\d[/-][0-1]?\d[/-]\d{2,4})\b",inv)
|
|
if m: inv=m.group(1); r["p_2a_inv_dt"]=self._format_date(m.group(2))
|
|
inv=re.sub(r"\s+[0-3]?\d[/-][0-1]?\d[/-]\d{2,4}$","",inv).strip()
|
|
if not re.fullmatch(r"\d{5,20}",inv) or (shipping_bill_no and inv==shipping_bill_no): continue
|
|
r["p_2a_inv_no"]=inv
|
|
if r.get("p_2a_inv_dt"): r["p_2a_inv_dt"]=self._format_date(r["p_2a_inv_dt"])
|
|
serial=self._clean(r.get("inv_sn",""))
|
|
if not re.fullmatch(r"\d{1,4}",serial) or serial==shipping_bill_no: r["inv_sn"]=""
|
|
if inv in seen:
|
|
for x in valid:
|
|
if x.get("p_2a_inv_no")==inv: self._merge_invoice_fragment(x,r); break
|
|
continue
|
|
seen.add(inv); valid.append(r)
|
|
for i,r in enumerate(valid,1):
|
|
if not re.fullmatch(r"\d{1,4}",self._clean(r.get("inv_sn",""))): r["inv_sn"]=i
|
|
result["invoice_details"]=valid
|
|
|
|
def _sanitize_item_details(self, result):
|
|
valid=[]; index={}
|
|
for raw in result.get("item_details",[]) or []:
|
|
r=dict(raw or {}); cth=re.sub(r"\D","",self._clean(r.get("p_3a_cth","")))
|
|
if not re.fullmatch(r"\d{6,10}",cth): continue
|
|
inv=self._clean(r.get("p_3a_invsno","")); item=self._clean(r.get("p_3a_itemsn",""))
|
|
if not re.fullmatch(r"\d{1,6}",inv): inv=""
|
|
if not re.fullmatch(r"\d{1,6}",item): item=""
|
|
commercial=any(self._clean(r.get(k,"")) for k in ("p_3a_item_desc","p_3a_qty","p_3a_uqc","p_3a_rate","p_3a_value","p_3a_fob"))
|
|
if not item and not commercial: continue
|
|
r.update({"p_3a_cth":cth,"p_3a_invsno":inv,"p_3a_itemsn":item,"doctype":"Item Detail"})
|
|
key=(inv,item,cth)
|
|
if key in index: self._merge_item_fragment(index[key],r)
|
|
else: index[key]=r; valid.append(r)
|
|
result["item_details"]=valid
|
|
|
|
def _sanitize_single_window(self, result):
|
|
valid=[]; seen=set()
|
|
for raw in result.get("single_window_declaration",[]) or []:
|
|
r=dict(raw or {}); info=self._clean(r.get("p_4d_info","")).upper(); q=self._clean(r.get("p_4d_qualifier","")).upper(); inv=self._clean(r.get("p_4d_invsn","")); item=self._clean(r.get("p_4d_itmsn",""))
|
|
if info not in {"CHR","DTY","ORC"} or q not in {"SQC","GCESS","RDT","DOO","EPT","STO"}: continue
|
|
if not re.fullmatch(r"\d{1,6}",inv) or not re.fullmatch(r"\d{1,6}",item): continue
|
|
r.update({"p_4d_invsn":inv,"p_4d_itmsn":item,"p_4d_info":info,"p_4d_qualifier":q,"p_4d_info_cd":self._clean(r.get("p_4d_info_cd","")),"p_4d_info_text":self._clean(r.get("p_4d_info_text","")),"p_4d_info_msr":self._clean(r.get("p_4d_info_msr","")),"p_4d_uqc":self._clean(r.get("p_4d_uqc","")).upper(),"doctype":"Single Window Declaration"})
|
|
key=(inv,item,info,q)
|
|
if key not in seen: seen.add(key); valid.append(r)
|
|
result["single_window_declaration"]=valid
|
|
|
|
def _sanitize_optional_child_sections(self, result):
|
|
for table in ("aa__dfia_licence_details","challan_details","drawback_and_rosl_claim","job_details","single_window_declaration_constituents","single_window_declaration_control","supporting_documents","ar4_details","third_party_details","item_manufacturerproducergrower_details","rodtep_details","container_details"):
|
|
result[table]=[r for r in (result.get(table) or []) if any(self._clean(v) for k,v in dict(r or {}).items() if k!="doctype")]
|
|
|
|
def _sanitize_final_output(self, result):
|
|
float_fields={"gwt","p_1c_fob_val","p_1c_com","p_1c_cess","p_1c_freight","p_1c_deductions","p_1c_insurance","p_1c_p_c","p_1c_discount","p_1c_duty","p_1d_dbk","p_1d_rodtep_amt","p_1d_igst_amt","p_1d_rosctle_amt","p_1d_cess_amt","p_1d_igst_val"}
|
|
int_fields={"inv","item","pkg","cont"}
|
|
for f in float_fields: result[f]=0.0 if self._float(result.get(f)) is None else self._float(result.get(f))
|
|
for f in int_fields: result[f]=0 if self._integer(result.get(f)) is None else self._integer(result.get(f))
|
|
for f in self.DATE_FIELDS: result[f]=self._format_date(result.get(f,"")) if self._clean(result.get(f,"")) else ""
|
|
for f in self.TIME_FIELDS: result[f]=self._format_time(result.get(f,"")) if self._clean(result.get(f,"")) else ""
|
|
if result.get("p_1b_ifsc_no") and not re.fullmatch(r"[A-Z]{4}0[A-Z0-9]{6}",str(result["p_1b_ifsc_no"]).upper()):
|
|
result["p_1b_ifsc_no"]=""
|
|
|
|
# These declarant fields must never receive spillover from adjacent
|
|
# labels/columns. Their actual values, when present, are normally a
|
|
# single compact token immediately associated with the label.
|
|
rbi_no = self._clean(result.get("p_1b_rbi_wr_no", ""))
|
|
if rbi_no and not re.fullmatch(r"[A-Z0-9][A-Z0-9./-]{3,29}", rbi_no.upper()):
|
|
result["p_1b_rbi_wr_no"] = ""
|
|
|
|
dbk_no = self._clean(result.get("p_1b_dbk_ac_no", ""))
|
|
if dbk_no and (" " in dbk_no or not re.fullmatch(r"[A-Z0-9./-]{4,30}", dbk_no.upper())):
|
|
result["p_1b_dbk_ac_no"] = ""
|
|
|
|
# RBI waiver date is populated only when a valid date is directly to
|
|
# the right of the RBI waiver label. This prevents Examination/LEO dates
|
|
# from leaking into the blank RBI block.
|
|
rbi_label = self._find_label(["RBI WAIVER NO", "RBI WAIVER DT", "RBI WAIVER DATE"])
|
|
if rbi_label:
|
|
rbi_date = self._value_right_same_row(rbi_label, validator=self._valid_date, max_distance=180)
|
|
result["p_1b_rbi_wr_dt"] = self._format_date(rbi_date) if rbi_date else ""
|
|
else:
|
|
result["p_1b_rbi_wr_dt"] = ""
|
|
for f in self.PARENT_FIELDS:
|
|
if f=="doctype": continue
|
|
result.setdefault(f,0.0 if f in float_fields else 0 if f in int_fields else "")
|
|
for t in self.CHILD_TABLES: result.setdefault(t,[])
|
|
|
|
# =========================================================
|
|
# ABSOLUTE FINAL CONTRACT CLEANUP
|
|
# =========================================================
|
|
|
|
def _final_contract_cleanup(self, result):
|
|
"""Apply the final conservative Frappe JSON contract cleanup."""
|
|
# System-managed fields are not extraction output.
|
|
result.pop("shipping_bill", None)
|
|
result.pop("naming_series", None)
|
|
|
|
# AEO must be a compact field value, never a layout/text spillover.
|
|
aeo = self._clean(result.get("p_1b_aeo", ""))
|
|
result["p_1b_aeo"] = aeo if self._valid_clean_aeo(aeo) else ""
|
|
|
|
# Recover CB name and buyer from their own document sections.
|
|
# CB NAME is authoritative only when recovered from its own Part-I
|
|
# visual cell. Never retain a generic nearby candidate here.
|
|
cb_name = self._recover_clean_cb_name()
|
|
result["p_1b_cb_name"] = cb_name if cb_name else ""
|
|
|
|
buyer_name = self._recover_clean_buyer_name()
|
|
if buyer_name:
|
|
result["buyer_name"] = buyer_name
|
|
elif not self._valid_party_name(self._clean(result.get("buyer_name", ""))):
|
|
result["buyer_name"] = ""
|
|
|
|
if buyer_name:
|
|
for invoice in result.get("invoice_details", []) or []:
|
|
if not self._clean(invoice.get("p_2b_buyer_name", "")):
|
|
invoice["p_2b_buyer_name"] = buyer_name
|
|
|
|
# Blank-only declarant fields remain blank unless a field-shaped value exists.
|
|
if not re.fullmatch(r"[A-Z]{4}0[A-Z0-9]{6}", self._clean(result.get("p_1b_ifsc_no", "")).upper()):
|
|
result["p_1b_ifsc_no"] = ""
|
|
|
|
for field in ("p_1b_dbk_ac_no", "p_1b_rbi_wr_no"):
|
|
value = self._clean(result.get(field, ""))
|
|
if not value or not re.fullmatch(r"[A-Z0-9][A-Z0-9./-]{3,29}", value.upper()):
|
|
result[field] = ""
|
|
|
|
# RBI waiver date is optional and must be sourced exclusively from
|
|
# the RBI WAIVER cell. Never retain a value extracted by a generic
|
|
# nearest-number/date rule.
|
|
result["p_1b_rbi_wr_dt"] = self._recover_clean_rbi_waiver_date()
|
|
|
|
# Manifest: only emit metadata actually represented by the source row.
|
|
# This prevents empty p_1e_noc / MAWB date / HAWB fields from appearing
|
|
# merely because the schema contains those optional fields.
|
|
clean_manifest = []
|
|
seen_manifest = set()
|
|
for raw in result.get("manifest_details", []) or []:
|
|
row = dict(raw or {})
|
|
mawb = self._clean(row.get("p_1e_mawb_no", ""))
|
|
cin = self._clean(row.get("p_1e_cin_no", "")).upper()
|
|
cin_dt = self._format_date(row.get("p_1e_cin_dt", ""))
|
|
site = self._clean(row.get("p_1e_cin_site_id", "")).upper()
|
|
if not re.fullmatch(r"\d{8,12}", mawb):
|
|
continue
|
|
if not re.fullmatch(r"[A-Z0-9]{16,24}", cin):
|
|
continue
|
|
if not re.fullmatch(r"\d{4}-\d{2}-\d{2}", cin_dt):
|
|
continue
|
|
if not re.fullmatch(r"[A-Z]{2}[A-Z]{3}\d", site):
|
|
continue
|
|
key = (mawb, cin)
|
|
if key in seen_manifest:
|
|
continue
|
|
seen_manifest.add(key)
|
|
clean_manifest.append({
|
|
"doctype": "Manifest Detail",
|
|
"p_1e_mawb_no": mawb,
|
|
"p_1e_cin_no": cin,
|
|
"p_1e_cin_dt": cin_dt,
|
|
"p_1e_cin_site_id": site,
|
|
})
|
|
result["manifest_details"] = clean_manifest
|
|
|
|
# Strict parent numeric types. These fields are Float in the supplied
|
|
# Customize Form schema and therefore must always be Python floats.
|
|
float_fields = {
|
|
"gwt", "p_1c_fob_val", "p_1c_com", "p_1c_cess", "p_1c_freight",
|
|
"p_1c_deductions", "p_1c_insurance", "p_1c_p_c", "p_1c_discount",
|
|
"p_1c_duty", "p_1d_dbk", "p_1d_rodtep_amt", "p_1d_igst_amt",
|
|
"p_1d_rosctle_amt", "p_1d_cess_amt", "p_1d_igst_val",
|
|
}
|
|
int_fields = {"inv", "item", "pkg", "cont"}
|
|
for field in float_fields:
|
|
parsed = self._float(result.get(field))
|
|
result[field] = 0.0 if parsed is None else float(parsed)
|
|
for field in int_fields:
|
|
parsed = self._integer(result.get(field))
|
|
result[field] = 0 if parsed is None else int(parsed)
|
|
|
|
# Final whitelist prevents accidental internal/system keys from leaking.
|
|
allowed = set(self.PARENT_FIELDS) | set(self.CHILD_TABLES) | {"doctype"}
|
|
for key in list(result.keys()):
|
|
if key not in allowed:
|
|
del result[key]
|
|
|
|
# Re-establish every canonical parent field except deliberately omitted
|
|
# Frappe-managed fields.
|
|
for field in self.PARENT_FIELDS:
|
|
if field == "doctype" or field in result:
|
|
continue
|
|
if field in float_fields:
|
|
result[field] = 0.0
|
|
elif field in int_fields:
|
|
result[field] = 0
|
|
elif field == "p_1a_re_exp":
|
|
result[field] = None
|
|
else:
|
|
result[field] = ""
|
|
|
|
@staticmethod
|
|
def _valid_party_name(value):
|
|
value = ShippingBillParser._clean(value)
|
|
if not value or len(value) > 160:
|
|
return False
|
|
|
|
upper = value.upper().strip()
|
|
|
|
# A party name must never be a numbered PDF label/header.
|
|
if re.match(r"^\s*\d{1,2}\s*[.:-]", upper):
|
|
return False
|
|
|
|
bad_tokens = (
|
|
"PORT OF ", "COUNTRY OF ", "STATE OF ", "VALUE", "FOB VALUE",
|
|
"DEDUCTIONS", "COMMISSION", "FREIGHT", "INSURANCE", "DISCOUNT",
|
|
"IGST", "ROSCTL", "RODTEP", "PART -", "SCAN QR", "GLOSSARY",
|
|
"SHIPPING BILL", "INV ITEM CONT", "GSTIN / TYPE", "FOREX BANK A/C",
|
|
"IFSC NO", "RBI WAIVER", "DBK BANK A/C", "CB NAME",
|
|
"AD CODE", "AEO", "TYPE", "STATUS", "CENTRAL BOARD",
|
|
)
|
|
if any(token in upper for token in bad_tokens):
|
|
return False
|
|
|
|
# A field label consisting mostly of punctuation/digits is not a party.
|
|
if re.fullmatch(r"[A-Z0-9 .:/&_-]{2,40}", upper) and not re.search(r"[A-Z]{3,}", upper):
|
|
return False
|
|
|
|
if re.search(r"\d{5,}", value):
|
|
return False
|
|
|
|
return bool(re.search(r"[A-Za-z]{3,}", value))
|
|
|
|
@staticmethod
|
|
def _valid_clean_aeo(value):
|
|
value = ShippingBillParser._clean(value)
|
|
if not value or len(value) > 50 or "\n" in value or "\r" in value:
|
|
return False
|
|
upper = value.upper().strip()
|
|
if re.match(r"^\s*\d{1,2}\s*[.:-]", upper):
|
|
return False
|
|
bad_tokens = (
|
|
"VALUE", "FOB", "DEDUCTIONS", "COMMISSION", "FREIGHT", "INSURANCE",
|
|
"DISCOUNT", "GSTIN", "FOREX", "PART -", "DECLAR", "INV ITEM",
|
|
"A M D P", "V U R", "CENTRAL", "STATUS", "IFSC", "RBI WAIVER",
|
|
)
|
|
if any(token in upper for token in bad_tokens):
|
|
return False
|
|
return bool(re.fullmatch(r"[A-Z0-9./_-]{6,40}", upper))
|
|
|
|
def _recover_clean_cb_name(self):
|
|
"""
|
|
Recover Customs Broker strictly from the Part-I CB NAME cell.
|
|
|
|
The previous implementation could accept a neighbouring numbered
|
|
label such as ``11. IFSC NO.`` because it only validated the text as
|
|
a generic party name. This implementation uses the actual visual
|
|
geometry of the CB NAME label:
|
|
|
|
CB NAME label
|
|
|
|
|
+--> same visual row, to the right
|
|
|
|
and stops before the next Part-I field column. If no valid value is
|
|
visually associated with CB NAME, the result is blank.
|
|
"""
|
|
|
|
# -------------------------------------------------------------
|
|
# 1. Visual word model — authoritative recovery
|
|
# -------------------------------------------------------------
|
|
for row in self.rows:
|
|
if row.get("_page") != 1:
|
|
continue
|
|
|
|
words = sorted(
|
|
row.get("words", []) or [],
|
|
key=lambda w: float(w.get("x0", 0)),
|
|
)
|
|
|
|
if not words:
|
|
continue
|
|
|
|
# Locate the actual CB NAME label, including the common
|
|
# pdfplumber split: ``5.CB`` + ``NAME``.
|
|
label_indices = []
|
|
for i, word in enumerate(words):
|
|
wt = self._clean(word.get("_text", ""))
|
|
if re.search(r"^5\.?CB$", wt, re.I):
|
|
if i + 1 < len(words) and re.fullmatch(r"NAME", self._clean(words[i + 1].get("_text", "")), re.I):
|
|
label_indices.append((i, i + 1))
|
|
elif re.fullmatch(r"CB\s*NAME", wt, re.I):
|
|
label_indices.append((i, i))
|
|
|
|
# Also support a combined label cell such as ``5.CB NAME``.
|
|
if not label_indices:
|
|
for i, word in enumerate(words):
|
|
wt = self._clean(word.get("_text", ""))
|
|
if re.search(r"(?:^|\s)CB\s*NAME(?:$|\s)", wt, re.I):
|
|
label_indices.append((i, i))
|
|
|
|
for start_idx, end_idx in label_indices:
|
|
label_x1 = max(
|
|
float(words[start_idx].get("x1", 0)),
|
|
float(words[end_idx].get("x1", 0)),
|
|
)
|
|
|
|
label_y = (
|
|
float(words[start_idx].get("y0", 0))
|
|
+ float(words[start_idx].get("y1", 0))
|
|
) / 2.0
|
|
|
|
# The next Part-I column starts around the next field label.
|
|
# We discover that boundary from labels rather than hardcoding
|
|
# a sample-PDF coordinate.
|
|
next_label_x = None
|
|
for candidate in words:
|
|
ct = self._clean(candidate.get("_text", ""))
|
|
if float(candidate.get("x0", 0)) <= label_x1:
|
|
continue
|
|
if re.match(r"^(?:8\.?GSTIN|9\.?FOREX|10\.?DBK|11\.?IFSC|6\.?AEO|4\.?RBI)", ct, re.I):
|
|
x = float(candidate.get("x0", 0))
|
|
if next_label_x is None or x < next_label_x:
|
|
next_label_x = x
|
|
|
|
# Collect only same-row words in the CB value column.
|
|
candidates = []
|
|
for candidate in words:
|
|
cx0 = float(candidate.get("x0", 0))
|
|
cy = (
|
|
float(candidate.get("y0", 0))
|
|
+ float(candidate.get("y1", 0))
|
|
) / 2.0
|
|
|
|
if cx0 <= label_x1 + 5:
|
|
continue
|
|
if abs(cy - label_y) > 4.5:
|
|
continue
|
|
if next_label_x is not None and cx0 >= next_label_x - 3:
|
|
continue
|
|
|
|
value = self._clean(candidate.get("_text", ""))
|
|
if not value or self._looks_like_field_label(value):
|
|
continue
|
|
candidates.append(candidate)
|
|
|
|
if candidates:
|
|
candidate_text = self._clean(
|
|
" ".join(
|
|
self._clean(w.get("_text", ""))
|
|
for w in sorted(candidates, key=lambda w: float(w.get("x0", 0)))
|
|
)
|
|
)
|
|
|
|
if self._valid_party_name(candidate_text):
|
|
return candidate_text
|
|
|
|
# -------------------------------------------------------------
|
|
# 2. Table fallback — only a cell immediately associated with
|
|
# the CB NAME header is accepted.
|
|
# -------------------------------------------------------------
|
|
for table in self.tables:
|
|
if table.get("page") != 1:
|
|
continue
|
|
|
|
rows = self._clean_table(table.get("rows", []))
|
|
for ridx, cells in enumerate(rows):
|
|
joined = self._clean(" ".join(cells)).upper()
|
|
if "CB NAME" not in joined and "5.CB NAME" not in joined:
|
|
continue
|
|
|
|
# Prefer the same row, then the immediately following row.
|
|
for candidate_cells in (
|
|
cells,
|
|
rows[ridx + 1] if ridx + 1 < len(rows) else [],
|
|
):
|
|
for cell in candidate_cells:
|
|
cell = self._clean(cell)
|
|
if not cell:
|
|
continue
|
|
if re.search(r"CB\s*NAME", cell, re.I):
|
|
continue
|
|
if self._looks_like_field_label(cell):
|
|
continue
|
|
if self._valid_party_name(cell):
|
|
return cell
|
|
|
|
return ""
|
|
|
|
def _looks_like_field_label(self, value):
|
|
value = self._clean(value)
|
|
if not value:
|
|
return True
|
|
|
|
upper = value.upper().strip()
|
|
|
|
if re.match(r"^\d{1,2}\s*[.:-]", upper):
|
|
return True
|
|
|
|
labels = (
|
|
"IFSC", "RBI WAIVER", "FOREX BANK", "DBK BANK", "GSTIN",
|
|
"AD CODE", "CB NAME", "AEO", "PORT CODE", "SB NO", "SB DATE",
|
|
"STATUS", "VALUE", "FOB VALUE", "FREIGHT", "INSURANCE",
|
|
"DISCOUNT", "DEDUCTIONS", "COMMISSION", "DUTY", "CESS",
|
|
"RODTEP", "ROSCTL", "GLOSSARY", "PART -", "CENTRAL BOARD",
|
|
)
|
|
|
|
return any(token in upper for token in labels)
|
|
|
|
def _recover_clean_buyer_name(self):
|
|
for row in self.rows:
|
|
if row.get("_page") != 2:
|
|
continue
|
|
text = self._clean(row.get("text", ""))
|
|
if not re.search(r"BUYER(?:'S|S)?\s*NAME\s*&\s*ADDRESS", text, re.I):
|
|
continue
|
|
row_y = float(row.get("_y", 0))
|
|
for candidate_row in self.rows:
|
|
if candidate_row.get("_page") != 2:
|
|
continue
|
|
cy = float(candidate_row.get("_y", 0))
|
|
if cy <= row_y or cy - row_y > 45:
|
|
continue
|
|
right_words = [w for w in candidate_row.get("words", []) if float(w.get("x0", 0)) > 280]
|
|
right_value = self._clean(" ".join(self._clean(w.get("_text", "")) for w in sorted(right_words, key=lambda w: float(w.get("x0", 0)))))
|
|
if self._valid_party_name(right_value):
|
|
return right_value
|
|
for table in self.tables:
|
|
if table.get("page") != 2:
|
|
continue
|
|
rows = self._clean_table(table.get("rows", []))
|
|
for ridx, cells in enumerate(rows):
|
|
joined = self._clean(" ".join(cells)).upper()
|
|
if "BUYER" not in joined or "NAME" not in joined or ridx + 1 >= len(rows):
|
|
continue
|
|
for cell in reversed(rows[ridx + 1]):
|
|
cell = self._clean(cell)
|
|
if self._valid_party_name(cell):
|
|
return cell
|
|
return ""
|
|
|
|
def _recover_clean_rbi_waiver_date(self):
|
|
"""
|
|
Recover RBI waiver date only from the RBI WAIVER field cell.
|
|
|
|
This field is optional. A nearby account number, examination date,
|
|
LEO date, or any other numeric token is never a valid fallback.
|
|
If a valid date is not visually associated with the RBI WAIVER label,
|
|
return an empty string.
|
|
"""
|
|
|
|
date_pattern = re.compile(
|
|
r"^(?:\d{1,2}[-/]?[A-Za-z]{3,4}[-/]?\d{2,4}|\d{1,2}/\d{1,2}/\d{2,4})$",
|
|
re.I,
|
|
)
|
|
|
|
for row in self.rows:
|
|
if row.get("_page") != 1:
|
|
continue
|
|
|
|
words = sorted(
|
|
row.get("words", []) or [],
|
|
key=lambda w: float(w.get("x0", 0)),
|
|
)
|
|
|
|
if not words:
|
|
continue
|
|
|
|
# Locate the complete RBI WAIVER label. In the sample PDF it is
|
|
# split into: 4.RBI | WAIVER | NO.& | DT.
|
|
label_words = []
|
|
for word in words:
|
|
wt = self._clean(word.get("_text", ""))
|
|
if re.search(r"RBI", wt, re.I):
|
|
label_words.append(word)
|
|
|
|
if not label_words:
|
|
continue
|
|
|
|
# Confirm that this is actually the RBI WAIVER field, not a
|
|
# glossary/reference occurrence.
|
|
row_text = self._clean(" ".join(self._clean(w.get("_text", "")) for w in words))
|
|
if not re.search(r"RBI\s+WAIVER", row_text, re.I):
|
|
continue
|
|
|
|
# Determine the right edge of the entire RBI WAIVER label.
|
|
# Include only contiguous label words until another known field
|
|
# column begins.
|
|
first_rbi = min(label_words, key=lambda w: float(w.get("x0", 0)))
|
|
start_x = float(first_rbi.get("x0", 0))
|
|
label_end_x = float(first_rbi.get("x1", 0))
|
|
|
|
for word in words:
|
|
if float(word.get("x0", 0)) < start_x:
|
|
continue
|
|
if float(word.get("x0", 0)) > label_end_x + 170:
|
|
break
|
|
wt = self._clean(word.get("_text", ""))
|
|
if re.search(r"^(?:RBI|WAIVER|NO\.?&?|DT\.?)$", wt, re.I) or re.search(r"^4\.?RBI$", wt, re.I):
|
|
label_end_x = max(label_end_x, float(word.get("x1", 0)))
|
|
|
|
# Find a date only to the right of the label, on the same visual
|
|
# row, within a bounded distance. Account numbers such as
|
|
# 51XXXX000 cannot pass this test.
|
|
candidates = []
|
|
label_y = (
|
|
float(first_rbi.get("y0", 0))
|
|
+ float(first_rbi.get("y1", 0))
|
|
) / 2.0
|
|
|
|
for word in words:
|
|
cx0 = float(word.get("x0", 0))
|
|
cy = (
|
|
float(word.get("y0", 0))
|
|
+ float(word.get("y1", 0))
|
|
) / 2.0
|
|
value = self._clean(word.get("_text", ""))
|
|
|
|
if cx0 <= label_end_x + 3:
|
|
continue
|
|
if cx0 - label_end_x > 220:
|
|
continue
|
|
if abs(cy - label_y) > 5:
|
|
continue
|
|
if not date_pattern.fullmatch(value):
|
|
continue
|
|
if not self._valid_date(value):
|
|
continue
|
|
|
|
candidates.append((cx0 - label_end_x, value))
|
|
|
|
if candidates:
|
|
candidates.sort(key=lambda item: item[0])
|
|
return self._format_date(candidates[0][1])
|
|
|
|
# No valid date directly associated with RBI WAIVER. This is the
|
|
# correct result for the supplied PDF.
|
|
return ""
|
|
|
|
def _empty_result(self):
|
|
result = {"doctype": "Shipping Bill"}
|
|
|
|
for field in self.PARENT_FIELDS:
|
|
if field == "doctype":
|
|
continue
|
|
|
|
if field in self.NUMBER_FIELDS:
|
|
if field in {
|
|
"gwt",
|
|
"p_1c_fob_val",
|
|
"p_1c_com",
|
|
"p_1c_freight",
|
|
"p_1c_deductions",
|
|
"p_1c_insurance",
|
|
"p_1c_p_c",
|
|
"p_1c_discount",
|
|
"p_1d_dbk",
|
|
"p_1d_rodtep_amt",
|
|
"p_1d_rosctle_amt",
|
|
}:
|
|
result[field] = 0.0
|
|
else:
|
|
result[field] = 0
|
|
elif field == "p_1a_re_exp":
|
|
result[field] = None
|
|
else:
|
|
result[field] = ""
|
|
|
|
for table in self.CHILD_TABLES:
|
|
result[table] = []
|
|
|
|
return result
|
|
|
|
# =========================================================
|
|
# WORD / ROW MODEL
|
|
# =========================================================
|
|
|
|
def _all_words(self):
|
|
words = []
|
|
|
|
for page_index, page in enumerate(self.pages, start=1):
|
|
for word in page.get("words", []) or []:
|
|
text = self._clean(word.get("text", ""))
|
|
if not text:
|
|
continue
|
|
|
|
item = dict(word)
|
|
item["_page"] = page_index
|
|
item["_text"] = text
|
|
item["_norm"] = self._norm(text)
|
|
|
|
item["x0"] = self._num(item.get("x0"))
|
|
item["x1"] = self._num(item.get("x1"))
|
|
item["y0"] = self._num(item.get("y0", item.get("top", 0)))
|
|
item["y1"] = self._num(item.get("y1", item.get("bottom", 0)))
|
|
|
|
words.append(item)
|
|
|
|
# The preferred source is pdfplumber words. If an older/newer analyzer
|
|
# supplies only layout text, build a lightweight word model from the
|
|
# layout lines so the parser remains backwards compatible.
|
|
if words:
|
|
return words
|
|
|
|
for page_index, page in enumerate(self.pages, start=1):
|
|
text = page.get("text", "") or ""
|
|
for line_no, line in enumerate(text.splitlines()):
|
|
if not line.strip():
|
|
continue
|
|
# Preserve layout columns approximately. This is a fallback only;
|
|
# real pdfplumber words are preferred whenever available.
|
|
for match in re.finditer(r"\S+", line):
|
|
value = match.group()
|
|
item = {
|
|
"text": value,
|
|
"x0": float(match.start()),
|
|
"x1": float(match.end()),
|
|
"y0": float(line_no * 10),
|
|
"y1": float(line_no * 10 + 8),
|
|
"_page": page_index,
|
|
"_text": value,
|
|
"_norm": self._norm(value),
|
|
}
|
|
words.append(item)
|
|
|
|
return words
|
|
|
|
def _build_rows(self, words):
|
|
rows = []
|
|
|
|
for word in sorted(
|
|
words,
|
|
key=lambda w: (
|
|
w["_page"],
|
|
w["y0"],
|
|
w["x0"],
|
|
),
|
|
):
|
|
target = None
|
|
|
|
for row in reversed(rows[-8:]):
|
|
if row["_page"] != word["_page"]:
|
|
continue
|
|
|
|
if abs(row["_y"] - word["y0"]) <= 3.5:
|
|
target = row
|
|
break
|
|
|
|
if target is None:
|
|
target = {
|
|
"_page": word["_page"],
|
|
"_y": word["y0"],
|
|
"words": [],
|
|
}
|
|
rows.append(target)
|
|
|
|
target["words"].append(word)
|
|
|
|
for row in rows:
|
|
row["words"].sort(key=lambda w: w["x0"])
|
|
row["text"] = " ".join(w["_text"] for w in row["words"])
|
|
|
|
return rows
|
|
|
|
def _group_words_into_rows(self, words, y_tolerance=3.5):
|
|
"""Return visual rows as lists of word dictionaries."""
|
|
if words is self.words:
|
|
return [row["words"] for row in self.rows if row.get("words")]
|
|
|
|
rows = []
|
|
for word in sorted(
|
|
words or [],
|
|
key=lambda w: (
|
|
w.get("_page", 0),
|
|
self._num(w.get("y0", w.get("top", 0))),
|
|
self._num(w.get("x0", 0)),
|
|
),
|
|
):
|
|
page = word.get("_page", 0)
|
|
y = self._num(word.get("y0", word.get("top", 0)))
|
|
target = None
|
|
|
|
for row in reversed(rows[-8:]):
|
|
if row["_page"] != page:
|
|
continue
|
|
if abs(row["_y"] - y) <= y_tolerance:
|
|
target = row
|
|
break
|
|
|
|
if target is None:
|
|
target = {
|
|
"_page": page,
|
|
"_y": y,
|
|
"words": [],
|
|
}
|
|
rows.append(target)
|
|
|
|
target["words"].append(word)
|
|
|
|
for row in rows:
|
|
row["words"].sort(key=lambda w: self._num(w.get("x0", 0)))
|
|
|
|
return [row["words"] for row in rows if row["words"]]
|
|
|
|
def _all_tables(self):
|
|
result = []
|
|
|
|
for page_index, page in enumerate(
|
|
self.pages,
|
|
start=1,
|
|
):
|
|
for table_index, table in enumerate(
|
|
page.get("tables", []) or [],
|
|
start=1,
|
|
):
|
|
result.append(
|
|
{
|
|
"page": page_index,
|
|
"index": table_index,
|
|
"rows": table,
|
|
}
|
|
)
|
|
|
|
return result
|
|
|
|
# =========================================================
|
|
# HEADER
|
|
# =========================================================
|
|
|
|
def _extract_header(self, result):
|
|
"""
|
|
Extract the Shipping Bill header using visual columns.
|
|
|
|
The header is a special case: INV / ITEM / CONT / PKG / G.WT
|
|
are presented as neighbouring columns. A generic nearest-number
|
|
search can therefore steal a value from another column.
|
|
"""
|
|
self._extract_header_column_values(result)
|
|
|
|
header_specs = {
|
|
"port_code": (["PORT CODE"], self._valid_port),
|
|
"shipping_bill_no": (
|
|
["SB NO", "SB NO."],
|
|
self._valid_sb_number,
|
|
),
|
|
"shipping_bill_date": (
|
|
["SB DATE"],
|
|
self._valid_date,
|
|
),
|
|
"cb_code": (
|
|
["CB CODE"],
|
|
self._valid_cb_code,
|
|
),
|
|
}
|
|
|
|
for field, (aliases, validator) in header_specs.items():
|
|
if result.get(field):
|
|
continue
|
|
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
continue
|
|
|
|
value = self._value_from_header_column(
|
|
label,
|
|
validator=validator,
|
|
)
|
|
|
|
if value is None:
|
|
value = self._value_below_column(
|
|
label,
|
|
validator=validator,
|
|
)
|
|
|
|
if value is None:
|
|
value = self._value_right_same_row(
|
|
label,
|
|
validator=validator,
|
|
)
|
|
|
|
if value is not None:
|
|
result[field] = value
|
|
self._trace(
|
|
field,
|
|
value,
|
|
label["_page"],
|
|
)
|
|
|
|
iec = self._find_iec()
|
|
if iec:
|
|
result["iec"] = iec
|
|
self._trace("iec", iec)
|
|
|
|
branch = self._find_iec_branch(iec)
|
|
if branch:
|
|
result["iec_branch_code"] = branch
|
|
self._trace("iec_branch_code", branch)
|
|
|
|
if not result["gwt_unit"]:
|
|
unit = self._near_unit(["G.WT", "GWT", "GROSS WEIGHT"])
|
|
if unit:
|
|
result["gwt_unit"] = unit
|
|
self._trace("gwt_unit", unit)
|
|
|
|
self._extract_count_columns(result)
|
|
|
|
def _extract_header_column_values(self, result):
|
|
"""
|
|
Extract INV / ITEM / CONT / PKG / G.WT from the first-page
|
|
header by identifying the label row first and then reading the
|
|
next visual row at the same X positions.
|
|
"""
|
|
first_words = [w for w in self.words if w["_page"] == 1]
|
|
|
|
specs = [
|
|
("inv", ["INV"], True),
|
|
("item", ["ITEM"], True),
|
|
("cont", ["CONT"], True),
|
|
("pkg", ["PKG"], True),
|
|
("gwt", ["G.WT", "GWT"], False),
|
|
]
|
|
|
|
for field, aliases, is_integer in specs:
|
|
if result.get(field) not in ("", None, 0):
|
|
continue
|
|
|
|
label = self._find_label_in_words(
|
|
first_words,
|
|
aliases,
|
|
)
|
|
|
|
if not label:
|
|
continue
|
|
|
|
validator = lambda x: self._integer(x) is not None if is_integer else self._float(x) is not None
|
|
|
|
value = self._value_from_header_column(
|
|
label,
|
|
validator=validator,
|
|
max_vertical_gap=70,
|
|
max_horizontal_distance=75,
|
|
)
|
|
|
|
if value is None:
|
|
continue
|
|
|
|
number = self._integer(value) if is_integer else self._float(value)
|
|
|
|
if number is None:
|
|
continue
|
|
|
|
result[field] = number
|
|
self._trace(
|
|
field,
|
|
number,
|
|
label["_page"],
|
|
)
|
|
|
|
def _recover_header_counts(self, result):
|
|
"""
|
|
Recover INV / ITEM / CONT / PKG / G.WT from the actual visual
|
|
header/value row.
|
|
|
|
The important difference from a nearest-number search is that
|
|
all header labels are first located on page 1, then a numeric
|
|
row is selected, and values are paired by their horizontal
|
|
column order.
|
|
|
|
This prevents:
|
|
PKG -> 1
|
|
G.WT -> 1
|
|
|
|
when the actual row is:
|
|
1 | 1 | 0 | 2 | 1143 | KGS
|
|
"""
|
|
page_words = [w for w in self.words if w.get("_page") == 1]
|
|
|
|
if not page_words:
|
|
return
|
|
|
|
specs = [
|
|
("inv", ["INV"]),
|
|
("item", ["ITEM"]),
|
|
("cont", ["CONT"]),
|
|
("pkg", ["PKG"]),
|
|
("gwt", ["G.WT", "GWT", "GROSS WEIGHT"]),
|
|
]
|
|
|
|
labels = []
|
|
|
|
for field, aliases in specs:
|
|
label = self._find_label_in_words(page_words, aliases)
|
|
if label:
|
|
labels.append((field, label))
|
|
|
|
if not labels:
|
|
return
|
|
|
|
labels.sort(
|
|
key=lambda item: (
|
|
item[1]["y0"],
|
|
item[1]["x0"],
|
|
)
|
|
)
|
|
|
|
# Header labels normally share the same visual line. Group labels
|
|
# by their Y coordinate so an unrelated "ITEM" elsewhere on page 1
|
|
# cannot participate in the counter mapping.
|
|
groups = []
|
|
|
|
for field, label in labels:
|
|
placed = False
|
|
for group in groups:
|
|
if abs(label["y0"] - group["y"]) <= 8:
|
|
group["labels"].append((field, label))
|
|
placed = True
|
|
break
|
|
|
|
if not placed:
|
|
groups.append(
|
|
{
|
|
"y": label["y0"],
|
|
"labels": [(field, label)],
|
|
}
|
|
)
|
|
|
|
# Prefer the group containing the greatest number of the five
|
|
# header fields.
|
|
groups.sort(
|
|
key=lambda group: len(group["labels"]),
|
|
reverse=True,
|
|
)
|
|
|
|
for group in groups:
|
|
group_labels = sorted(
|
|
group["labels"],
|
|
key=lambda item: item[1]["x0"],
|
|
)
|
|
|
|
if len(group_labels) < 2:
|
|
continue
|
|
|
|
label_y1 = max(label["y1"] for _, label in group_labels)
|
|
|
|
# Find candidate rows below the header. A row is considered a
|
|
# counter row when it contains several numeric tokens.
|
|
candidate_rows = [
|
|
row
|
|
for row in self.rows
|
|
if row.get("_page") == 1 and row["_y"] >= label_y1 and row["_y"] - label_y1 <= 100
|
|
]
|
|
|
|
best = None
|
|
|
|
for row in candidate_rows:
|
|
numeric_words = []
|
|
|
|
for word in row["words"]:
|
|
value = self._clean(word.get("_text", ""))
|
|
|
|
# Ignore decimal fragments and ordinary text.
|
|
if not re.fullmatch(
|
|
r"\d+(?:[,.]\d+)?",
|
|
value,
|
|
):
|
|
continue
|
|
|
|
numeric_words.append(word)
|
|
|
|
if len(numeric_words) < 2:
|
|
continue
|
|
|
|
numeric_words.sort(key=lambda w: w["x0"])
|
|
|
|
# Pair each header with the closest numeric token in X.
|
|
pairs = []
|
|
used = set()
|
|
|
|
for field, label in group_labels:
|
|
label_center = (label["x0"] + label["x1"]) / 2
|
|
|
|
available = [
|
|
(index, word) for index, word in enumerate(numeric_words) if index not in used
|
|
]
|
|
|
|
if not available:
|
|
continue
|
|
|
|
index, word = min(
|
|
available,
|
|
key=lambda item: abs((item[1]["x0"] + item[1]["x1"]) / 2 - label_center),
|
|
)
|
|
|
|
used.add(index)
|
|
pairs.append(
|
|
(
|
|
field,
|
|
word,
|
|
abs((word["x0"] + word["x1"]) / 2 - label_center),
|
|
)
|
|
)
|
|
|
|
if len(pairs) < 2:
|
|
continue
|
|
|
|
score = sum(distance for _, _, distance in pairs)
|
|
|
|
if best is None or score < best[0]:
|
|
best = (score, pairs)
|
|
|
|
if best is None:
|
|
continue
|
|
|
|
# The header's visual sequence is ordered. If both the label
|
|
# and value columns are sorted left-to-right, pairing should
|
|
# also be monotonic. Reject a crossed mapping.
|
|
pairs = best[1]
|
|
pairs.sort(key=lambda item: next(label["x0"] for f, label in group_labels if f == item[0]))
|
|
|
|
value_xs = [(word["x0"] + word["x1"]) / 2 for _, word, _ in pairs]
|
|
|
|
if value_xs != sorted(value_xs):
|
|
continue
|
|
|
|
for field, word, _ in pairs:
|
|
text = self._clean(word["_text"])
|
|
|
|
if field == "gwt":
|
|
value = self._float(text)
|
|
# Gross weight must not be confused with the package,
|
|
# invoice, item, or container counters.
|
|
if value is None or value <= 0:
|
|
continue
|
|
if value < 10:
|
|
continue
|
|
|
|
result[field] = value
|
|
self._trace(field, value, 1)
|
|
|
|
else:
|
|
value = self._integer(text)
|
|
if value is None or value < 0:
|
|
continue
|
|
|
|
# Never overwrite a valid non-zero extraction unless
|
|
# the current value is missing/zero.
|
|
if result.get(field) not in ("", None, 0, 0.0):
|
|
continue
|
|
|
|
result[field] = value
|
|
self._trace(field, value, 1)
|
|
|
|
# If a valid G.WT was found, the unit can usually be recovered
|
|
# from the nearest KGS/KG token.
|
|
if result.get("gwt") and not result.get("gwt_unit"):
|
|
self._recover_weight_unit()
|
|
|
|
if result.get("pkg") not in ("", None, 0, 0.0) or result.get("gwt") not in ("", None, 0, 0.0):
|
|
return
|
|
|
|
def _recover_weight_unit(self):
|
|
for row in self.rows:
|
|
if row.get("_page") != 1:
|
|
continue
|
|
|
|
for word in row["words"]:
|
|
text = self._clean(word.get("_text", "")).upper().rstrip(".")
|
|
if text in {"KGS", "KG"}:
|
|
self._trace("gwt_unit", text, 1)
|
|
return
|
|
|
|
def _recover_gwt_pkg_from_text(self, result):
|
|
"""
|
|
Recover GWT and PKG from the Shipping Bill package/weight summary.
|
|
|
|
Reference PDF layout can expose the summary as:
|
|
|
|
400099 2 KGS 1143
|
|
|
|
Here:
|
|
400099 -> exporter pincode / neighbouring field
|
|
2 -> packages
|
|
KGS -> weight unit
|
|
1143 -> gross weight
|
|
|
|
Therefore the parser must NOT use the first number near KGS.
|
|
The number immediately AFTER KGS is the gross weight in this
|
|
layout, while the number immediately BEFORE KGS is the package
|
|
count.
|
|
|
|
The method also supports conventional:
|
|
2 PKG 1143 KGS
|
|
G.WT 1143 KGS
|
|
GWT 1143 KGS
|
|
"""
|
|
text = str(self.full_text or "")
|
|
|
|
lines = [self._clean(line) for line in text.splitlines() if self._clean(line)]
|
|
|
|
gwt_candidates = []
|
|
pkg_candidates = []
|
|
|
|
# =========================================================
|
|
# 1. SHIPPING BILL SUMMARY PATTERN
|
|
# =========================================================
|
|
#
|
|
# Critical reference-form pattern:
|
|
#
|
|
# 400099 2 KGS 1143
|
|
#
|
|
# Do NOT take 400099.
|
|
# Take:
|
|
# package = 2
|
|
# unit = KGS
|
|
# gwt = 1143
|
|
#
|
|
summary_pattern = re.compile(
|
|
r"(?<!\d)"
|
|
r"(\d{6})"
|
|
r"\s+"
|
|
r"(\d{1,5})"
|
|
r"\s+"
|
|
r"(KGS?|KG)"
|
|
r"\s+"
|
|
r"(\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d+(?:\.\d+)?)"
|
|
r"(?!\d)",
|
|
re.I,
|
|
)
|
|
|
|
for line in lines:
|
|
for match in summary_pattern.finditer(line):
|
|
pincode = match.group(1)
|
|
pkg_raw = match.group(2)
|
|
gwt_raw = match.group(4).replace(",", "")
|
|
|
|
# Six digits before the package count is intentionally
|
|
# treated as a neighbouring/pincode field.
|
|
if not re.fullmatch(r"\d{6}", pincode):
|
|
continue
|
|
|
|
try:
|
|
pkg = int(pkg_raw)
|
|
gwt = float(gwt_raw)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
|
|
if not (0 <= pkg <= 100000):
|
|
continue
|
|
|
|
if not (0 < gwt <= 10_000_000):
|
|
continue
|
|
|
|
gwt_candidates.append((3000, gwt))
|
|
pkg_candidates.append((3000, pkg))
|
|
|
|
# Same pattern can be split across two layout lines.
|
|
if not gwt_candidates:
|
|
joined_text = " ".join(lines)
|
|
|
|
for match in summary_pattern.finditer(joined_text):
|
|
try:
|
|
pkg = int(match.group(2))
|
|
gwt = float(match.group(4).replace(",", ""))
|
|
except (TypeError, ValueError):
|
|
continue
|
|
|
|
if 0 <= pkg <= 100000 and 0 < gwt <= 10_000_000:
|
|
gwt_candidates.append((2900, gwt))
|
|
pkg_candidates.append((2900, pkg))
|
|
|
|
# =========================================================
|
|
# 2. EXPLICIT G.WT / GWT / GROSS WEIGHT
|
|
# =========================================================
|
|
for index, line in enumerate(lines):
|
|
if not re.search(
|
|
r"\bG\s*\.?\s*W\s*\.?\s*T\b|\bGWT\b|GROSS\s+WEIGHT",
|
|
line,
|
|
re.I,
|
|
):
|
|
continue
|
|
|
|
nearby = line
|
|
|
|
if index + 1 < len(lines):
|
|
nearby += " " + lines[index + 1]
|
|
|
|
# Prefer number immediately followed by KGS.
|
|
for match in re.finditer(
|
|
r"(?<![A-Z0-9])"
|
|
r"(\d{1,3}(?:,\d{3})+(?:\.\d+)?|\d+(?:\.\d+)?)"
|
|
r"\s*(KGS?|KG)\b",
|
|
nearby,
|
|
re.I,
|
|
):
|
|
try:
|
|
value = float(match.group(1).replace(",", ""))
|
|
except (TypeError, ValueError):
|
|
continue
|
|
|
|
if 0 < value <= 10_000_000:
|
|
gwt_candidates.append((2500, value))
|
|
|
|
# =========================================================
|
|
# 3. VISUAL ROW: NUMBER + KGS + NUMBER
|
|
# =========================================================
|
|
#
|
|
# Handles pdfplumber extraction where:
|
|
# 400099 | 2 | KGS | 1143
|
|
#
|
|
# becomes four separate words.
|
|
for row in self._group_words_into_rows(self.words):
|
|
row_words = sorted(
|
|
row,
|
|
key=lambda w: float(w.get("x0", 0)),
|
|
)
|
|
|
|
cells = [self._clean(w.get("_text", "")) for w in row_words]
|
|
|
|
for i in range(len(cells) - 2):
|
|
# Pattern: package | KGS | GWT
|
|
if not re.fullmatch(r"\d{1,5}", cells[i]):
|
|
continue
|
|
|
|
if cells[i + 1].upper().rstrip(".") not in {"KG", "KGS"}:
|
|
continue
|
|
|
|
if not re.fullmatch(
|
|
r"\d+(?:\.\d+)?",
|
|
cells[i + 2].replace(",", ""),
|
|
):
|
|
continue
|
|
|
|
try:
|
|
pkg = int(cells[i])
|
|
gwt = float(cells[i + 2].replace(",", ""))
|
|
except (TypeError, ValueError):
|
|
continue
|
|
|
|
if 0 <= pkg <= 100000 and 0 < gwt <= 10_000_000:
|
|
gwt_candidates.append((2400, gwt))
|
|
pkg_candidates.append((2400, pkg))
|
|
|
|
# Pattern: pincode | package | KGS | GWT
|
|
for i in range(len(cells) - 3):
|
|
if not re.fullmatch(r"\d{6}", cells[i]):
|
|
continue
|
|
|
|
if not re.fullmatch(r"\d{1,5}", cells[i + 1]):
|
|
continue
|
|
|
|
if cells[i + 2].upper().rstrip(".") not in {"KG", "KGS"}:
|
|
continue
|
|
|
|
if not re.fullmatch(
|
|
r"\d+(?:\.\d+)?",
|
|
cells[i + 3].replace(",", ""),
|
|
):
|
|
continue
|
|
|
|
try:
|
|
pkg = int(cells[i + 1])
|
|
gwt = float(cells[i + 3].replace(",", ""))
|
|
except (TypeError, ValueError):
|
|
continue
|
|
|
|
if 0 <= pkg <= 100000 and 0 < gwt <= 10_000_000:
|
|
gwt_candidates.append((3500, gwt))
|
|
pkg_candidates.append((3500, pkg))
|
|
|
|
# =========================================================
|
|
# 4. APPLY STRONGEST CANDIDATES
|
|
# =========================================================
|
|
if gwt_candidates:
|
|
gwt_candidates.sort(
|
|
key=lambda item: (item[0], item[1]),
|
|
reverse=True,
|
|
)
|
|
|
|
result["gwt"] = float(gwt_candidates[0][1])
|
|
result["gwt_unit"] = "KGS"
|
|
|
|
self._trace("gwt", result["gwt"])
|
|
self._trace("gwt_unit", "KGS")
|
|
|
|
if pkg_candidates:
|
|
pkg_candidates.sort(
|
|
key=lambda item: (item[0], -item[1]),
|
|
reverse=True,
|
|
)
|
|
|
|
result["pkg"] = int(pkg_candidates[0][1])
|
|
self._trace("pkg", result["pkg"])
|
|
|
|
if result.get("gwt"):
|
|
result["gwt_unit"] = "KGS"
|
|
|
|
def _extract_count_columns(self, result):
|
|
"""
|
|
Compatibility fallback for header layouts where the labels are
|
|
not found as expected. It uses exact label positions and never
|
|
accepts another known label as a value.
|
|
"""
|
|
specs = {
|
|
"inv": ["INV"],
|
|
"item": ["ITEM"],
|
|
"cont": ["CONT"],
|
|
}
|
|
|
|
for field, aliases in specs.items():
|
|
if result.get(field) not in ("", None, 0):
|
|
continue
|
|
|
|
label = self._find_label_in_words(
|
|
[w for w in self.words if w["_page"] == 1],
|
|
aliases,
|
|
)
|
|
|
|
if not label:
|
|
continue
|
|
|
|
value = self._value_from_header_column(
|
|
label,
|
|
validator=lambda x: self._integer(x) is not None,
|
|
max_vertical_gap=70,
|
|
max_horizontal_distance=75,
|
|
)
|
|
|
|
if value is None:
|
|
continue
|
|
|
|
number = self._integer(value)
|
|
if number is None:
|
|
continue
|
|
|
|
result[field] = number
|
|
self._trace(
|
|
field,
|
|
number,
|
|
label["_page"],
|
|
)
|
|
|
|
# =========================================================
|
|
# STATUS FLAGS
|
|
# =========================================================
|
|
|
|
def _extract_statuses(self, result):
|
|
status_aliases = {
|
|
"p_1a_assess": ["ASSESS"],
|
|
"p_1a_dbk": ["DBK"],
|
|
"p_1a_re_exp": ["RE-EXP", "RE EXP", "REEXP"],
|
|
"p_1a_exam": ["EXMN", "EXAM"],
|
|
"p_1a_rodtp": ["RODTP", "RODTEP"],
|
|
"p_1a_lut": ["LUT"],
|
|
"p_1a_job": ["JOBBING", "JOB"],
|
|
"p_1a_licence": ["LICENCE", "LICENSE"],
|
|
"p_1a_dfrc": ["DFRC"],
|
|
}
|
|
|
|
for field, aliases in status_aliases.items():
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
continue
|
|
|
|
token = self._nearest_status_token(label)
|
|
if token is None:
|
|
continue
|
|
|
|
if field == "p_1a_re_exp":
|
|
result[field] = token
|
|
else:
|
|
result[field] = token
|
|
|
|
self._trace(field, token, label["_page"])
|
|
|
|
mode = self._find_label(["MODE"])
|
|
if mode:
|
|
value = self._nearest_mode(mode)
|
|
if value:
|
|
result["p_1a_mode"] = value
|
|
self._trace(
|
|
"p_1a_mode",
|
|
value,
|
|
mode["_page"],
|
|
)
|
|
|
|
def _nearest_status_token(self, label):
|
|
candidates = []
|
|
|
|
label_center = (label["x0"] + label["x1"]) / 2
|
|
|
|
for word in self.words:
|
|
if word["_page"] != label["_page"]:
|
|
continue
|
|
|
|
value = word["_text"].upper().strip(".,:-")
|
|
|
|
if value not in {"Y", "N"}:
|
|
continue
|
|
|
|
if word["y0"] < label["y1"] - 2:
|
|
continue
|
|
|
|
dy = word["y0"] - label["y1"]
|
|
if dy > 70:
|
|
continue
|
|
|
|
word_center = (word["x0"] + word["x1"]) / 2
|
|
dx = abs(word_center - label_center)
|
|
|
|
# A status flag belongs to the same visual column.
|
|
if dx > 75:
|
|
continue
|
|
|
|
candidates.append(
|
|
(
|
|
dy + dx * 0.5,
|
|
word,
|
|
)
|
|
)
|
|
|
|
if not candidates:
|
|
return None
|
|
|
|
candidates.sort(key=lambda x: x[0])
|
|
return candidates[0][1]["_text"].upper()
|
|
|
|
def _nearest_mode(self, label):
|
|
allowed = {
|
|
"AIR",
|
|
"SEA",
|
|
"ROAD",
|
|
"RAIL",
|
|
"COURIER",
|
|
"POST",
|
|
}
|
|
|
|
candidates = []
|
|
|
|
for word in self.words:
|
|
if word["_page"] != label["_page"]:
|
|
continue
|
|
|
|
value = word["_text"].upper().strip()
|
|
if value not in allowed:
|
|
continue
|
|
|
|
if word["y0"] < label["y0"] - 5:
|
|
continue
|
|
|
|
dy = word["y0"] - label["y0"]
|
|
if dy > 80:
|
|
continue
|
|
|
|
center1 = (label["x0"] + label["x1"]) / 2
|
|
center2 = (word["x0"] + word["x1"]) / 2
|
|
|
|
candidates.append(
|
|
(
|
|
dy + abs(center1 - center2) * 0.35,
|
|
value,
|
|
)
|
|
)
|
|
|
|
if not candidates:
|
|
return None
|
|
|
|
candidates.sort()
|
|
return candidates[0][1]
|
|
|
|
def _value_same_row_region(self, label, validator=None, max_x_gap=420):
|
|
"""
|
|
Return the value block to the right of a label on the same visual row.
|
|
|
|
Shipping Bill Part-I is a two-column grid. The label and its value are
|
|
often on the same PDF row, not on separate rows.
|
|
"""
|
|
page = label["_page"]
|
|
ly = (label["y0"] + label["y1"]) / 2
|
|
candidates = [
|
|
w
|
|
for w in self.words
|
|
if w["_page"] == page
|
|
and abs(((w["y0"] + w["y1"]) / 2) - ly) <= 4
|
|
and w["x0"] >= label["x1"]
|
|
and w["x0"] - label["x1"] <= max_x_gap
|
|
]
|
|
|
|
if not candidates:
|
|
return None
|
|
|
|
candidates.sort(key=lambda w: w["x0"])
|
|
selected = []
|
|
for word in candidates:
|
|
value = self._clean(word["_text"])
|
|
if self._label_is_value(value):
|
|
if selected:
|
|
break
|
|
continue
|
|
if validator and not validator(value):
|
|
continue
|
|
selected.append(word)
|
|
|
|
if not selected:
|
|
return None
|
|
|
|
# For codes/numbers one token is enough. For locations/names, retain
|
|
# contiguous words until the next label boundary.
|
|
return self._clean(" ".join(w["_text"] for w in selected))
|
|
|
|
# =========================================================
|
|
# TARGETED TABLE / ROW OVERRIDES
|
|
# =========================================================
|
|
|
|
def _table_row_text(self, row):
|
|
return self._clean(" ".join(str(c or "") for c in row))
|
|
|
|
@staticmethod
|
|
def _compact_label(value):
|
|
return re.sub(r"[^a-z0-9]+", "", str(value or "").lower())
|
|
|
|
def _find_table_row(self, rows, *needles):
|
|
needles = [self._compact_label(x) for x in needles if x]
|
|
for index, row in enumerate(rows):
|
|
text = self._compact_label(self._table_row_text(row))
|
|
if all(n in text for n in needles):
|
|
return index
|
|
return None
|
|
|
|
def _cell_after_label(self, row, label_parts, stop_parts=()):
|
|
"""
|
|
Extract the complete value following a table label.
|
|
|
|
Important:
|
|
pdfplumber may split a port into multiple cells, for example:
|
|
|
|
PORT OF LOADING | I | NBOM4 | (Mumbai (Ex Bombay))
|
|
|
|
The parser therefore collects adjacent cells instead of returning
|
|
the first cell only.
|
|
"""
|
|
labels = [self._compact_label(x) for x in label_parts]
|
|
stops = [self._compact_label(x) for x in stop_parts]
|
|
|
|
for index, cell in enumerate(row):
|
|
text = self._clean(cell)
|
|
|
|
if not text:
|
|
continue
|
|
|
|
compact_text = self._compact_label(text)
|
|
|
|
matched_label = None
|
|
matched_end = None
|
|
|
|
for raw_label, compact_label in zip(label_parts, labels):
|
|
if not compact_label:
|
|
continue
|
|
|
|
# Normal exact/substring match.
|
|
position = compact_text.find(compact_label)
|
|
|
|
if position >= 0:
|
|
matched_label = raw_label
|
|
matched_end = position + len(compact_label)
|
|
break
|
|
|
|
if matched_label is None:
|
|
continue
|
|
|
|
# IMPORTANT:
|
|
# compact_text may have removed spaces/punctuation, so do not
|
|
# use matched_end directly against the original text. First try
|
|
# the original label with regex.
|
|
tail = ""
|
|
|
|
for raw_label in label_parts:
|
|
match = re.search(
|
|
re.escape(raw_label),
|
|
text,
|
|
flags=re.IGNORECASE,
|
|
)
|
|
|
|
if match:
|
|
tail = self._clean(text[match.end() :].lstrip(" :.-"))
|
|
break
|
|
|
|
# If the label was detected only after normalization, there may
|
|
# be no exact original-label match. In that case the label cell
|
|
# itself contains no value and extraction starts in the next cell.
|
|
if tail and not self._label_is_value(tail):
|
|
parts = [tail]
|
|
else:
|
|
parts = []
|
|
|
|
# Collect all adjacent cells until a known following label.
|
|
for next_cell in row[index + 1 :]:
|
|
value = self._clean(next_cell)
|
|
|
|
if not value:
|
|
continue
|
|
|
|
compact_value = self._compact_label(value)
|
|
|
|
# Stop at the next known field label.
|
|
if any(stop and (stop in compact_value or compact_value.startswith(stop)) for stop in stops):
|
|
break
|
|
|
|
# Do not stop on a one-character I/J fragment. Those can be
|
|
# the first physical piece of INBOM4/JED.
|
|
if self._label_is_value(value):
|
|
if len(value) == 1 and value.upper() in {"I", "J"}:
|
|
parts.append(value)
|
|
continue
|
|
|
|
break
|
|
|
|
parts.append(value)
|
|
|
|
value = self._join_location_cells(parts)
|
|
|
|
# Never return a clipped one-character location.
|
|
compact_result = re.sub(
|
|
r"[^A-Za-z0-9]+",
|
|
"",
|
|
value,
|
|
)
|
|
|
|
if len(compact_result) >= 3:
|
|
return value
|
|
|
|
return ""
|
|
|
|
def _join_location_cells(self, parts):
|
|
"""
|
|
Join adjacent pdfplumber table cells into one location value.
|
|
|
|
Some Shipping Bill PDFs split a port code across cells, for example:
|
|
|
|
I | NBOM4 | (Mumbai (Ex Bombay))
|
|
|
|
or:
|
|
|
|
J | ED | (JEDDAH )
|
|
|
|
The first fragment must be joined to the following cell instead
|
|
of being returned as a clipped value.
|
|
"""
|
|
result = ""
|
|
|
|
for part in parts or []:
|
|
part = self._clean(str(part or ""))
|
|
if not part:
|
|
continue
|
|
|
|
if not result:
|
|
result = part
|
|
continue
|
|
|
|
# Join single-letter port fragments directly to the next
|
|
# alphanumeric fragment: I + NBOM4 -> INBOM4.
|
|
if len(result) == 1 and result.upper() in {"I", "J"} and re.match(r"^[A-Za-z0-9]", part):
|
|
result += part
|
|
continue
|
|
|
|
# Join a code fragment directly when the previous part is an
|
|
# incomplete alphanumeric port code.
|
|
if (
|
|
re.fullmatch(r"[A-Za-z]{1,4}", result)
|
|
and re.fullmatch(r"[A-Za-z0-9]{1,8}", part)
|
|
and len(result) + len(part) <= 8
|
|
):
|
|
result += part
|
|
continue
|
|
|
|
result += " " + part
|
|
|
|
return self._clean(result)
|
|
|
|
def _extract_locations_from_tables(self, result):
|
|
"""
|
|
Extract Part-I locations from complete table spans.
|
|
|
|
pdfplumber can split a port into narrow cells such as:
|
|
I | NBOM4 | (Mumbai (Ex Bombay))
|
|
or:
|
|
J | ED | (JEDDAH )
|
|
|
|
The parser therefore evaluates the complete span instead of accepting
|
|
the first cell as the value.
|
|
"""
|
|
location_specs = [
|
|
(
|
|
"port_of_loading",
|
|
("12.PORT OF LOADING", "PORT OF LOADING"),
|
|
("13.COUNTRY OF FINAL", "13.COUNTRY OF FINALDESTINATION"),
|
|
),
|
|
(
|
|
"cntry_of_finaldstn",
|
|
(
|
|
"13.COUNTRY OF FINALDESTINATION",
|
|
"13.COUNTRY OF FINALDESTINATIO",
|
|
"COUNTRY OF FINALDESTINATION",
|
|
"COUNTRY OF FINAL DESTINATION",
|
|
),
|
|
("14.STATE OF ORIGIN", "STATE OF ORIGIN"),
|
|
),
|
|
(
|
|
"state_of_origin",
|
|
("14.STATE OF ORIGIN", "STATE OF ORIGIN"),
|
|
("15.PORT OF FINAL DESTINATION", "PORT OF FINAL DESTINATION"),
|
|
),
|
|
(
|
|
"port_of_finaldstn",
|
|
("15.PORT OF FINAL DESTINATION", "PORT OF FINAL DESTINATION"),
|
|
("16.PORT OF DISCHARGE", "PORT OF DISCHARGE"),
|
|
),
|
|
(
|
|
"port_of_discharge",
|
|
("16.PORT OF DISCHARGE", "PORT OF DISCHARGE"),
|
|
("17.COUNTRY OF DISCHARGE", "COUNTRY OF DISCHARGE"),
|
|
),
|
|
("cntry_of_discharge", ("17.COUNTRY OF DISCHARGE", "COUNTRY OF DISCHARGE"), ()),
|
|
]
|
|
|
|
for table_info in self.tables:
|
|
if table_info["page"] != 1:
|
|
continue
|
|
|
|
rows = self._clean_table(table_info["rows"])
|
|
|
|
for field, labels, stops in location_specs:
|
|
best_value = ""
|
|
|
|
for row in rows:
|
|
value = self._cell_after_label(row, labels, stops)
|
|
if not value:
|
|
continue
|
|
|
|
value = self._clean_location(value)
|
|
if field == "cntry_of_finaldstn":
|
|
value = self._repair_country_text(value)
|
|
|
|
if not self._valid_location_candidate(field, value):
|
|
continue
|
|
|
|
# Never accept a clipped I/J port fragment.
|
|
if (
|
|
field
|
|
in {
|
|
"port_of_loading",
|
|
"port_of_finaldstn",
|
|
"port_of_discharge",
|
|
}
|
|
and len(re.sub(r"[^A-Za-z0-9]+", "", value)) < 3
|
|
):
|
|
continue
|
|
|
|
if len(value) > len(best_value):
|
|
best_value = value
|
|
|
|
if best_value:
|
|
result[field] = best_value
|
|
self._trace(field, best_value, table_info["page"])
|
|
|
|
# Repair incomplete table values from the visual word model.
|
|
for field in (
|
|
"port_of_loading",
|
|
"port_of_finaldstn",
|
|
"port_of_discharge",
|
|
"cntry_of_finaldstn",
|
|
"cntry_of_discharge",
|
|
"state_of_origin",
|
|
):
|
|
current = self._clean(result.get(field, ""))
|
|
|
|
if current and self._location_value_is_complete(field, current):
|
|
continue
|
|
|
|
value = self._extract_complete_location_from_words(field)
|
|
if value:
|
|
result[field] = value
|
|
self._trace(field, value, 1)
|
|
|
|
def _valid_location_candidate(self, field, value):
|
|
value = self._clean(value)
|
|
if not value:
|
|
return False
|
|
|
|
if field in {
|
|
"port_of_loading",
|
|
"port_of_finaldstn",
|
|
"port_of_discharge",
|
|
}:
|
|
return self._validate_location_field(value)
|
|
|
|
if field == "state_of_origin":
|
|
return self._validate_state_field(value)
|
|
|
|
if field.startswith("cntry_"):
|
|
return self._validate_country_field(value)
|
|
|
|
return True
|
|
|
|
def _location_value_is_complete(self, field, value):
|
|
value = self._clean(value)
|
|
if not value:
|
|
return False
|
|
|
|
if field in {
|
|
"port_of_loading",
|
|
"port_of_finaldstn",
|
|
"port_of_discharge",
|
|
}:
|
|
compact = re.sub(r"[^A-Za-z0-9]+", "", value)
|
|
if len(compact) < 3:
|
|
return False
|
|
if re.fullmatch(r"[IJ]", value.upper()):
|
|
return False
|
|
|
|
return True
|
|
|
|
def _extract_complete_location_from_words(self, field):
|
|
"""
|
|
Reconstruct a location by joining the complete visual word span.
|
|
Unlike nearest-word extraction, this never returns only the first
|
|
fragment of a horizontally split value.
|
|
"""
|
|
aliases = {
|
|
"port_of_loading": ["PORT OF LOADING", "12.PORT OF LOADING"],
|
|
"cntry_of_finaldstn": [
|
|
"COUNTRY OF FINAL DESTINATION",
|
|
"COUNTRY OF FINALDESTINATION",
|
|
"13.COUNTRY OF FINALDESTINATION",
|
|
],
|
|
"state_of_origin": ["STATE OF ORIGIN", "14.STATE OF ORIGIN"],
|
|
"port_of_finaldstn": [
|
|
"PORT OF FINAL DESTINATION",
|
|
"15.PORT OF FINAL DESTINATION",
|
|
],
|
|
"port_of_discharge": [
|
|
"PORT OF DISCHARGE",
|
|
"16.PORT OF DISCHARGE",
|
|
],
|
|
"cntry_of_discharge": [
|
|
"COUNTRY OF DISCHARGE",
|
|
"17.COUNTRY OF DISCHARGE",
|
|
],
|
|
}
|
|
|
|
label = self._find_label(aliases.get(field, []))
|
|
if not label:
|
|
return ""
|
|
|
|
page = label["_page"]
|
|
label_x1 = label["x1"]
|
|
label_y = (label["y0"] + label["y1"]) / 2
|
|
|
|
candidate_rows = [
|
|
row
|
|
for row in self.rows
|
|
if row["_page"] == page and row["_y"] >= label["y0"] - 2 and row["_y"] <= label["y1"] + 110
|
|
]
|
|
|
|
candidates = []
|
|
|
|
for row in candidate_rows:
|
|
words = sorted(row["words"], key=lambda w: w["x0"])
|
|
|
|
# Same-row value: wide allowance.
|
|
right = [w for w in words if w["x0"] >= label_x1 - 8 and w["x0"] <= label_x1 + 550]
|
|
if right:
|
|
text = self._join_location_words(right)
|
|
if self._valid_location_candidate(field, text):
|
|
candidates.append((abs(row["_y"] - label_y), -len(text), text))
|
|
|
|
# Below-label value: same broad horizontal region.
|
|
below = [w for w in words if w["x0"] >= label["x0"] - 40 and w["x0"] <= label_x1 + 550]
|
|
if below:
|
|
text = self._join_location_words(below)
|
|
if self._valid_location_candidate(field, text):
|
|
candidates.append((abs(row["_y"] - label["y1"]), -len(text), text))
|
|
|
|
if not candidates:
|
|
return ""
|
|
|
|
candidates.sort(key=lambda x: (x[0], x[1]))
|
|
value = self._clean(candidates[0][2])
|
|
|
|
if field == "cntry_of_finaldstn":
|
|
value = self._repair_country_text(value)
|
|
|
|
return value
|
|
|
|
@staticmethod
|
|
def _join_location_words(words):
|
|
parts = []
|
|
|
|
for word in sorted(words, key=lambda w: w["x0"]):
|
|
text = str(word.get("_text", "")).strip()
|
|
if not text:
|
|
continue
|
|
|
|
if not parts:
|
|
parts.append(text)
|
|
continue
|
|
|
|
if len(parts[-1]) == 1 and parts[-1].upper() in {"I", "J"} and re.match(r"^[A-Za-z0-9]", text):
|
|
parts[-1] += text
|
|
else:
|
|
parts.append(text)
|
|
|
|
return " ".join(parts)
|
|
|
|
def _extract_parties_from_tables(self, result):
|
|
"""Prefer bounded Part-I table cells for party names.
|
|
|
|
Exporter name is exactly the first value row under its label. The
|
|
previous block collector could continue into later Part-I sections and
|
|
capture ``MANUFACTURER/PRODUCER/GROWER DETAILS``.
|
|
"""
|
|
for table_info in self.tables:
|
|
if table_info["page"] != 1:
|
|
continue
|
|
rows = self._clean_table(table_info["rows"])
|
|
for row_index, row in enumerate(rows):
|
|
row_text = self._table_row_text(row).upper()
|
|
if "EXPORTER'S NAME & ADDRESS" in row_text or "1.EXPORTER'S NAME & ADDRESS" in row_text:
|
|
if row_index + 1 < len(rows):
|
|
next_row = rows[row_index + 1]
|
|
value = self._cell_after_label(
|
|
next_row, ("EXPORTER'S NAME & ADDRESS", "1.EXPORTER'S NAME & ADDRESS")
|
|
)
|
|
if not value:
|
|
# In the clean table the value is simply the first
|
|
# non-empty cell on the exporter side.
|
|
value = self._first_nonempty_cell(next_row, left_half=True)
|
|
if value:
|
|
result["p_1b_exporter_name"] = self._clean_party_value(
|
|
"p_1b_exporter_name", value
|
|
)
|
|
self._trace(
|
|
"p_1b_exporter_name", result["p_1b_exporter_name"], table_info["page"]
|
|
)
|
|
|
|
# Address lines are bounded to the next explicit CB/AD section,
|
|
# never the entire remaining page.
|
|
addr = []
|
|
for rr in rows[row_index + 2 : row_index + 7]:
|
|
text = self._first_nonempty_cell(rr, left_half=True)
|
|
if not text:
|
|
continue
|
|
if any(
|
|
x in text.upper()
|
|
for x in (
|
|
"CB NAME",
|
|
"AD CODE",
|
|
"FOREX BANK",
|
|
"MANUFACTURER",
|
|
"PRODUCER",
|
|
"GROWER",
|
|
)
|
|
):
|
|
break
|
|
addr.append(text)
|
|
if addr:
|
|
result["p_1b_exporter_address"] = "\n".join(addr)
|
|
self._trace(
|
|
"p_1b_exporter_address", result["p_1b_exporter_address"], table_info["page"]
|
|
)
|
|
|
|
if "CONSIGNEE NAME & ADDRESS" in row_text:
|
|
if row_index + 1 < len(rows):
|
|
next_row = rows[row_index + 1]
|
|
value = self._first_nonempty_cell(next_row, left_half=False)
|
|
if value:
|
|
result["p_1b_consignee_name"] = self._clean_party_value(
|
|
"p_1b_consignee_name", value
|
|
)
|
|
self._trace(
|
|
"p_1b_consignee_name", result["p_1b_consignee_name"], table_info["page"]
|
|
)
|
|
|
|
addr = []
|
|
for rr in rows[row_index + 2 : row_index + 6]:
|
|
text = self._first_nonempty_cell(rr, left_half=False)
|
|
if not text:
|
|
continue
|
|
if any(
|
|
x in text.upper()
|
|
for x in ("GSTIN", "FOREX BANK", "AD CODE", "MANUFACTURER", "PRODUCER", "GROWER")
|
|
):
|
|
break
|
|
addr.append(text)
|
|
if addr:
|
|
result["p_1b_consignee_address"] = "\n".join(addr)
|
|
self._trace(
|
|
"p_1b_consignee_address", result["p_1b_consignee_address"], table_info["page"]
|
|
)
|
|
|
|
def _first_nonempty_cell(self, row, left_half=True):
|
|
values = [self._clean(x) for x in row if self._clean(x)]
|
|
if not values:
|
|
return ""
|
|
# For the two-column Part-I party block, exporter occupies the left
|
|
# side and consignee occupies the right side. When the extractor has
|
|
# already collapsed empty cells, use text cues to choose the side.
|
|
if left_half:
|
|
return values[0]
|
|
return values[-1]
|
|
|
|
def _extract_known_section_tables(self, result):
|
|
"""Directly map the two tables whose section headers identify them.
|
|
|
|
This is deliberately section-driven rather than classifier-driven. A
|
|
Shipping Bill's pdfplumber table can be fragmented into many pieces;
|
|
requiring the whole canonical header in one extracted table is therefore
|
|
too strict.
|
|
"""
|
|
invoice_records = []
|
|
item_records = []
|
|
|
|
for table_info in self.tables:
|
|
rows = self._clean_table(table_info["rows"])
|
|
if not rows:
|
|
continue
|
|
|
|
joined = self._compact_label(" ".join(self._table_row_text(r) for r in rows[:12]))
|
|
|
|
if "partiiinvoicedetails" in joined:
|
|
recs = self._map_invoice_section_table(rows)
|
|
invoice_records.extend(recs)
|
|
|
|
if "partiiiitemdetails" in joined:
|
|
recs = self._map_item_section_table(rows)
|
|
item_records.extend(recs)
|
|
|
|
if invoice_records:
|
|
result["invoice_details"] = self._merge_invoice_records(
|
|
result["invoice_details"] + invoice_records
|
|
)
|
|
|
|
if item_records:
|
|
result["item_details"] = self._merge_records_by_key(
|
|
result["item_details"] + item_records,
|
|
["p_3a_invsno", "p_3a_itemsn"],
|
|
)
|
|
result["item_details"] = self._collapse_item_continuations(result["item_details"])
|
|
|
|
def _map_invoice_section_table(self, rows):
|
|
header_index = self._find_table_row(rows, "s.no", "invoice no")
|
|
if header_index is None:
|
|
return []
|
|
|
|
headers = self._make_headers(rows[header_index])
|
|
record = {"doctype": "Invoice Detail"}
|
|
|
|
data_rows = rows[header_index + 1 :]
|
|
# Main invoice row is the first row containing a numeric serial and
|
|
# invoice/date information.
|
|
main = None
|
|
for row in data_rows[:4]:
|
|
text = self._table_row_text(row)
|
|
if re.search(r"\b\d{1,2}\b", text) and re.search(r"\d{5,}.*\d{1,2}[/-]\d{1,2}[/-]\d{2,4}", text):
|
|
main = row
|
|
break
|
|
if main is None and data_rows:
|
|
main = data_rows[0]
|
|
|
|
if main:
|
|
values = self._row_values(headers, main)
|
|
self._set_first(record, "inv_sn", values, ["s no", "sno", "sn", "serial", "1sno"])
|
|
self._set_first(record, "p_2a_inv_no", values, ["invoice no", "inv no", "2invoice no dt"])
|
|
if not record.get("p_2a_inv_no"):
|
|
m = re.search(r"\b(\d{5,})\s+([0-3]?\d[/-][0-1]?\d[/-]\d{2,4})\b", self._table_row_text(main))
|
|
if m:
|
|
record["p_2a_inv_no"] = m.group(1)
|
|
record["p_2a_inv_dt"] = self._format_date(m.group(2))
|
|
self._set_first(record, "p_2a_inv_dt", values, ["invoice date", "inv dt", "date"], date=True)
|
|
self._set_first(
|
|
record,
|
|
"p_2a_ad_code",
|
|
values,
|
|
["ad code", "6ad code"],
|
|
)
|
|
self._set_first(record, "p_2a_invterm", values, ["invterm", "term", "7invterm"])
|
|
|
|
# If the main row contains combined invoice/date, split it now.
|
|
if record.get("p_2a_inv_no"):
|
|
m = re.search(r"\b(\d{5,})\s+([0-3]?\d[/-][0-1]?\d[/-]\d{2,4})\b", record["p_2a_inv_no"])
|
|
if m:
|
|
record["p_2a_inv_no"] = m.group(1)
|
|
record["p_2a_inv_dt"] = self._format_date(m.group(2))
|
|
|
|
# Party block.
|
|
party_idx = self._find_table_row(rows, "exporter", "name", "address")
|
|
if party_idx is not None:
|
|
for rr in rows[party_idx + 1 : party_idx + 6]:
|
|
vals = [self._clean(x) for x in rr]
|
|
if not any(vals):
|
|
continue
|
|
left = self._first_nonempty_cell(rr, True)
|
|
right = self._first_nonempty_cell(rr, False)
|
|
if left and not record.get("p_2b_exporter_name"):
|
|
record["p_2b_exporter_name"] = left
|
|
if right and right != left and not record.get("p_2b_buyer_name"):
|
|
record["p_2b_buyer_name"] = right
|
|
if record.get("p_2b_exporter_name") and record.get("p_2b_buyer_name"):
|
|
break
|
|
|
|
# Value block.
|
|
value_idx = self._find_table_row(rows, "invoice value", "fob value", "exchange rate")
|
|
if value_idx is not None and value_idx + 1 < len(rows):
|
|
vh = self._make_headers(rows[value_idx])
|
|
vv = self._row_values(vh, rows[value_idx + 1])
|
|
self._set_first(record, "p_2c_invoice_value", vv, ["invoice value"], number=True)
|
|
self._set_first(record, "p_2c_fob_val", vv, ["fob value"], number=True)
|
|
self._set_first(record, "p_2c_freight", vv, ["freight"], number=True)
|
|
self._set_first(record, "p_2c_insurance", vv, ["insurance"], number=True)
|
|
self._set_first(record, "p_2c_discount", vv, ["discount"], number=True)
|
|
self._set_first(record, "p_2c_commison", vv, ["commison", "commission"], number=True)
|
|
self._set_first(record, "p_2c_deduct", vv, ["deduct"], number=True)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_exchng_rate_desc",
|
|
vv,
|
|
["exchange rate"],
|
|
)
|
|
if value_idx + 2 < len(rows):
|
|
curr = self._row_values(vh, rows[value_idx + 2])
|
|
self._set_first(
|
|
record,
|
|
"p_2c_invoice_curr",
|
|
curr,
|
|
["invoice value"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_fob_curr",
|
|
curr,
|
|
["fob value"],
|
|
)
|
|
|
|
if not record.get("inv_sn"):
|
|
record["inv_sn"] = 1 if record.get("p_2a_inv_no") else ""
|
|
return [record] if record.get("p_2a_inv_no") else []
|
|
|
|
def _map_item_section_table(self, rows):
|
|
header_index = self._find_table_row(rows, "invsn", "itemsn", "hs cd", "description")
|
|
if header_index is None:
|
|
return []
|
|
|
|
headers = self._make_headers(rows[header_index])
|
|
record = {"doctype": "Item Detail"}
|
|
data_index = header_index + 1
|
|
if data_index >= len(rows):
|
|
return []
|
|
|
|
# First row carries the core item values.
|
|
core = rows[data_index]
|
|
values = self._row_values(headers, core)
|
|
self._set_first(record, "p_3a_invsno", values, ["invsn", "1invsn", "inv s no", "invoice"])
|
|
self._set_first(record, "p_3a_itemsn", values, ["itemsn", "2itemsn", "item s no", "item"])
|
|
self._set_first(record, "p_3a_cth", values, ["hscd", "3hs cd", "cth", "hs code"])
|
|
self._set_first(record, "p_3a_item_desc", values, ["description", "4description", "item description"])
|
|
self._set_first(record, "p_3a_qty", values, ["quantity", "4quantity", "qty"])
|
|
self._set_first(record, "p_3a_uqc", values, ["uqc", "5uqc", "unit"])
|
|
self._set_first(record, "p_3a_rate", values, ["rate", "6rate"])
|
|
self._set_first(record, "p_3a_value", values, ["value", "7value"])
|
|
self._set_first(record, "p_3a_fob", values, ["fob", "9fob"])
|
|
self._set_first(record, "p_3a_pmv", values, ["pmv", "10pmv"])
|
|
|
|
# Explicitly resolve FOB/PMV columns when pdfplumber splits the
|
|
# header into narrow cells.
|
|
for idx, header in enumerate(headers):
|
|
h = self._compact_label(header)
|
|
if idx >= len(core):
|
|
continue
|
|
value = self._clean(core[idx])
|
|
if self._float(value) is None:
|
|
continue
|
|
if "fob" in h and not record.get("p_3a_fob"):
|
|
record["p_3a_fob"] = value
|
|
if "pmv" in h and not record.get("p_3a_pmv"):
|
|
record["p_3a_pmv"] = value
|
|
|
|
# Continuation rows belong to the same item. Append only text that is
|
|
# in the description column; do not absorb footer/next-section text.
|
|
desc_idx = next((i for i, h in enumerate(headers) if "description" in self._compact_label(h)), None)
|
|
if desc_idx is not None:
|
|
desc_parts = (
|
|
[str(core[desc_idx]).strip()] if desc_idx < len(core) and self._clean(core[desc_idx]) else []
|
|
)
|
|
for rr in rows[data_index + 1 :]:
|
|
rtxt = self._table_row_text(rr)
|
|
if self._compact_label(rtxt).startswith(("11dutyamt", "19scheme", "24ptabroad", "glossary")):
|
|
break
|
|
if desc_idx < len(rr):
|
|
cell = self._clean(rr[desc_idx])
|
|
if cell and not self._label_is_value(cell):
|
|
desc_parts.append(cell)
|
|
if desc_parts:
|
|
record["p_3a_item_desc"] = " ".join(desc_parts)
|
|
|
|
# Additional item attribute blocks are separate header/value rows.
|
|
# These tables are often split into very narrow cells, so use the
|
|
# label row to locate the following value row and then validate the
|
|
# candidate by the field type instead of relying on exact cell width.
|
|
for i, row in enumerate(rows):
|
|
text = self._compact_label(self._table_row_text(row))
|
|
if "11dutyamt" in text and i + 1 < len(rows):
|
|
value_text = self._table_row_text(rows[i + 1])
|
|
m = re.search(r"\b(N|Y)\b", value_text, re.I)
|
|
if m:
|
|
record["p_3a_dbk_claimed"] = m.group(1).upper()
|
|
m = re.search(r"\bP\s*LUT\b|\bLUT\b", value_text, re.I)
|
|
if m:
|
|
record["p_3a_igststat"] = "LUT"
|
|
m = re.search(r"\b\d{2}\b", value_text)
|
|
if m:
|
|
record["p_3a_schcod"] = m.group(0)
|
|
|
|
if "19schemedescription" in text and i + 1 < len(rows):
|
|
value_text = self._table_row_text(rows[i + 1])
|
|
# The first non-label text is the scheme description. Preserve it
|
|
# as printed; do not infer or rewrite it.
|
|
parts = [self._clean(x) for x in rows[i + 1] if self._clean(x)]
|
|
if parts:
|
|
record["p_3a_scheme_desc"] = parts[0]
|
|
m = re.search(r"\b(\d+(?:\.\d+)?)\b", value_text)
|
|
if m:
|
|
record["p_3a_sqc_mst"] = m.group(1)
|
|
m = re.search(r"\b(NOS|KGS|SET|PCS|UNIT|INR|USD)\b", value_text, re.I)
|
|
if m:
|
|
record["p_3a_sqc_uqc"] = m.group(1).upper()
|
|
m = re.search(
|
|
r"\b(?:NOS|KGS|SET|PCS|UNIT)\s+(Maharashtra|[A-Z][A-Za-z]+)\b", value_text, re.I
|
|
)
|
|
if m:
|
|
record["p_3a_state_of_origin"] = m.group(1)
|
|
m = re.search(r"\bMUMBAI\b", value_text, re.I)
|
|
if m:
|
|
record["p_3a_district_of_origin"] = "MUMBAI"
|
|
|
|
if "24ptabroad" in text and i + 1 < len(rows):
|
|
value_text = self._table_row_text(rows[i + 1])
|
|
parts = [self._clean(x) for x in rows[i + 1] if self._clean(x)]
|
|
if parts:
|
|
record["p_3a_pt_abroad"] = parts[0]
|
|
if "0 INR" in value_text.upper() or re.search(r"\b0\s*INR\b", value_text, re.I):
|
|
record["p_3a_comp_cess"] = 0.0
|
|
m = re.search(r"\b[A-Z]{3}\d{3}\b", value_text)
|
|
if m:
|
|
record["p_3a_end_use"] = m.group(0)
|
|
if re.search(r"\bY\b", value_text):
|
|
record["p_3a_benefit_availd"] = "Y"
|
|
if re.search(r"\bNo\b", value_text, re.I):
|
|
record["p_3a_reward_benefit"] = "No"
|
|
if re.search(r"\bN\b", value_text):
|
|
record["p_3a_third_party_item"] = "N"
|
|
|
|
if not record.get("p_3a_invsno"):
|
|
record["p_3a_invsno"] = "1" if record.get("p_3a_itemsn") else ""
|
|
if not record.get("p_3a_itemsn"):
|
|
m = re.search(r"\b(\d+)\s+(\d+)\s+\d{8}\b", self._table_row_text(core))
|
|
if m:
|
|
record["p_3a_invsno"], record["p_3a_itemsn"] = m.group(1), m.group(2)
|
|
return [record] if record.get("p_3a_itemsn") and record.get("p_3a_cth") else []
|
|
|
|
# =========================================================
|
|
# LOCATIONS
|
|
# =========================================================
|
|
|
|
def _extract_locations(self, result):
|
|
specs = {
|
|
"port_of_loading": (
|
|
["PORT OF LOADING"],
|
|
self._validate_location_field,
|
|
),
|
|
"state_of_origin": (
|
|
["STATE OF ORIGIN"],
|
|
self._validate_state_field,
|
|
),
|
|
"port_of_finaldstn": (
|
|
["PORT OF FINAL DESTINATION"],
|
|
self._validate_location_field,
|
|
),
|
|
"port_of_discharge": (
|
|
["PORT OF DISCHARGE"],
|
|
self._validate_location_field,
|
|
),
|
|
"cntry_of_finaldstn": (
|
|
[
|
|
"COUNTRY OF FINAL DESTINATION",
|
|
"COUNTRY OF FINALDESTINATION",
|
|
"COUNTRY OF FINALDESTINATIO",
|
|
],
|
|
self._validate_country_field,
|
|
),
|
|
"cntry_of_discharge": (
|
|
["COUNTRY OF DISCHARGE"],
|
|
self._validate_country_field,
|
|
),
|
|
}
|
|
|
|
for field, (aliases, validator) in specs.items():
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
continue
|
|
|
|
# Location values can be split into several narrow PDF words/cells.
|
|
# Read the complete visual span first; do not validate each word
|
|
# independently because ``I`` and ``J`` are valid one-character
|
|
# fragments of INBOM4/JED.
|
|
value = self._location_same_row_span(
|
|
label,
|
|
validator=validator,
|
|
max_x_gap=900,
|
|
)
|
|
|
|
if not value:
|
|
value = self._value_same_row_region(
|
|
label,
|
|
validator=validator,
|
|
max_x_gap=900,
|
|
)
|
|
|
|
if not value:
|
|
value = self._extract_block_after_label(
|
|
label,
|
|
validator=validator,
|
|
max_rows=3,
|
|
max_y_gap=100,
|
|
)
|
|
|
|
if not value:
|
|
continue
|
|
|
|
value = self._clean_location(value)
|
|
|
|
if value:
|
|
result[field] = value
|
|
self._trace(
|
|
field,
|
|
value,
|
|
label["_page"],
|
|
)
|
|
|
|
def _location_same_row_span(self, label, validator=None, max_x_gap=900):
|
|
"""Read a complete location span on the label's visual row.
|
|
|
|
Unlike nearest-word extraction, this method concatenates adjacent
|
|
words. It is deliberately permissive horizontally because PDF text
|
|
can be split into tiny fragments by the producer.
|
|
"""
|
|
page = label["_page"]
|
|
label_y = (label["y0"] + label["y1"]) / 2
|
|
words = [
|
|
w
|
|
for w in self.words
|
|
if w["_page"] == page
|
|
and abs(((w["y0"] + w["y1"]) / 2) - label_y) <= 5
|
|
and w["x0"] >= label["x1"]
|
|
and w["x0"] - label["x1"] <= max_x_gap
|
|
]
|
|
words.sort(key=lambda w: w["x0"])
|
|
|
|
if not words:
|
|
return ""
|
|
|
|
parts = []
|
|
for w in words:
|
|
text = self._clean(w["_text"])
|
|
if not text:
|
|
continue
|
|
|
|
# Do not absorb the next logical field label.
|
|
if self._label_is_value(text):
|
|
if parts:
|
|
break
|
|
continue
|
|
|
|
# If a multi-word known label starts here, stop before it.
|
|
remaining = self._clean(" ".join(x["_text"] for x in words[len(parts) :]))
|
|
if self._looks_like_location_stop(text):
|
|
if parts:
|
|
break
|
|
|
|
if validator and not validator(text):
|
|
# A location can be split into fragments. Keep alphabetic
|
|
# fragments instead of discarding them individually.
|
|
if not re.search(r"[A-Za-z]", text):
|
|
continue
|
|
|
|
parts.append(text)
|
|
|
|
value = self._clean(" ".join(parts))
|
|
return value if value and (not validator or validator(value)) else value
|
|
|
|
def _looks_like_location_stop(self, value):
|
|
compact = self._compact_label(value)
|
|
stops = {
|
|
"stateoforigin",
|
|
"portofloading",
|
|
"portoffinaldestination",
|
|
"portofdischarge",
|
|
"countryoffinaldestination",
|
|
"countryofdischarge",
|
|
"exportersnameaddress",
|
|
"consigneenameaddress",
|
|
}
|
|
return compact in stops
|
|
|
|
def _extract_block_after_label(
|
|
self,
|
|
label,
|
|
validator=None,
|
|
max_rows=3,
|
|
max_y_gap=90,
|
|
):
|
|
"""
|
|
Collect text from the same visual column below a label until
|
|
another field label begins.
|
|
|
|
This prevents PORT OF LOADING from absorbing:
|
|
PORT OF DISCHARGE
|
|
COUNTRY OF DISCHARGE
|
|
"""
|
|
label_center = (label["x0"] + label["x1"]) / 2
|
|
|
|
rows = [
|
|
row
|
|
for row in self.rows
|
|
if row["_page"] == label["_page"]
|
|
and row["_y"] >= label["y1"]
|
|
and row["_y"] - label["y1"] <= max_y_gap
|
|
]
|
|
|
|
rows.sort(key=lambda r: r["_y"])
|
|
|
|
collected = []
|
|
|
|
for row in rows:
|
|
row_text = self._clean(row["text"])
|
|
|
|
if not row_text:
|
|
continue
|
|
|
|
if self._label_is_value(row_text):
|
|
if collected:
|
|
break
|
|
continue
|
|
|
|
selected = []
|
|
|
|
for word in row["words"]:
|
|
center = (word["x0"] + word["x1"]) / 2
|
|
|
|
if abs(center - label_center) > 180:
|
|
continue
|
|
|
|
value = self._clean(word["_text"])
|
|
|
|
if self._label_is_value(value):
|
|
continue
|
|
|
|
if validator and not validator(value):
|
|
continue
|
|
|
|
selected.append(word)
|
|
|
|
if not selected:
|
|
continue
|
|
|
|
selected.sort(key=lambda w: w["x0"])
|
|
|
|
text = " ".join(w["_text"] for w in selected).strip()
|
|
|
|
if text:
|
|
collected.append(text)
|
|
|
|
if len(collected) >= max_rows:
|
|
break
|
|
|
|
return "\n".join(collected)
|
|
|
|
@staticmethod
|
|
def _validate_location_field(value):
|
|
value = ShippingBillParser._clean(value)
|
|
|
|
if not value:
|
|
return False
|
|
|
|
norm = ShippingBillParser._norm(value)
|
|
|
|
if norm in {
|
|
"port of loading",
|
|
"port of final destination",
|
|
"port of discharge",
|
|
"country of final destination",
|
|
"country of discharge",
|
|
"state of origin",
|
|
}:
|
|
return False
|
|
|
|
if re.fullmatch(r"[YN]", value.upper()):
|
|
return False
|
|
|
|
if re.fullmatch(r"[IJ]", value.upper()):
|
|
return False
|
|
|
|
if len(re.sub(r"[^A-Za-z0-9]+", "", value)) < 3:
|
|
return False
|
|
|
|
if re.fullmatch(r"\d+(?:\.\d+)?", value):
|
|
return False
|
|
|
|
return bool(re.search(r"[A-Za-z]", value))
|
|
|
|
@staticmethod
|
|
def _validate_country_field(value):
|
|
value = ShippingBillParser._clean(value)
|
|
norm = ShippingBillParser._norm(value)
|
|
|
|
if not value:
|
|
return False
|
|
|
|
if any(
|
|
token in norm
|
|
for token in (
|
|
"port of",
|
|
"country of",
|
|
"state of",
|
|
"exporter",
|
|
"consignee",
|
|
)
|
|
):
|
|
return False
|
|
|
|
if re.fullmatch(r"[YN]", value.upper()):
|
|
return False
|
|
|
|
return bool(re.search(r"[A-Za-z]", value))
|
|
|
|
@staticmethod
|
|
def _validate_state_field(value):
|
|
value = ShippingBillParser._clean(value)
|
|
norm = ShippingBillParser._norm(value)
|
|
|
|
if not value or len(value) < 3:
|
|
return False
|
|
|
|
if norm in {
|
|
"state of origin",
|
|
"district",
|
|
"copy",
|
|
}:
|
|
return False
|
|
|
|
if any(
|
|
token in norm
|
|
for token in (
|
|
"port of",
|
|
"country of",
|
|
"exporter",
|
|
"consignee",
|
|
)
|
|
):
|
|
return False
|
|
|
|
return bool(re.search(r"[A-Za-z]", value))
|
|
|
|
# =========================================================
|
|
# PARTIES
|
|
# =========================================================
|
|
|
|
def _extract_parties(self, result):
|
|
party_specs = {
|
|
"p_1b_exporter_name": [
|
|
"EXPORTER'S NAME & ADDRESS",
|
|
"1.EXPORTER'S NAME & ADDRESS",
|
|
"EXPORTER NAME & ADDRESS",
|
|
],
|
|
"p_1b_consignee_name": [
|
|
"CONSIGNEE NAME & ADDRESS",
|
|
"7.CONSIGNEE NAME & ADDRESS",
|
|
],
|
|
"p_1b_cb_name": ["CB NAME"],
|
|
"p_1b_type": ["TYPE"],
|
|
"p_1b_gstin": ["GSTIN", "GSTIN / TYPE"],
|
|
"p_1b_ad_code": ["AD CODE"],
|
|
"p_1b_forex_ac_no": [
|
|
"FOREX BANK A/C NO",
|
|
"FOREX BANK A/C NO.",
|
|
],
|
|
"p_1b_dbk_ac_no": ["DBK BANK A/C NO", "DBK BANK A/C NO."],
|
|
"p_1b_aeo": ["AEO"],
|
|
"p_1b_ifsc_no": ["IFSC NO", "IFSC NO."],
|
|
"p_1b_rbi_wr_no": ["RBI WAIVER NO", "RBI WAIVER NO."],
|
|
"p_1b_rbi_wr_dt": ["RBI WAIVER DT", "RBI WAIVER DATE"],
|
|
}
|
|
|
|
for field, aliases in party_specs.items():
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
continue
|
|
|
|
validator = self._party_validator(field)
|
|
|
|
value = self._value_in_visual_region(
|
|
label,
|
|
other_labels=list(party_specs.values()),
|
|
validator=validator,
|
|
max_y_gap=130,
|
|
)
|
|
|
|
if value:
|
|
result[field] = self._clean_party_value(
|
|
field,
|
|
value,
|
|
)
|
|
self._trace(
|
|
field,
|
|
result[field],
|
|
label["_page"],
|
|
)
|
|
|
|
self._extract_party_blocks(result)
|
|
|
|
def _party_validator(self, field):
|
|
if field == "p_1b_gstin":
|
|
return lambda x: bool(
|
|
re.search(
|
|
r"\b\d{2}[A-Z]{5}\d{4}[A-Z][A-Z0-9]Z[A-Z0-9]\b",
|
|
x.upper(),
|
|
)
|
|
)
|
|
|
|
if field == "p_1b_ad_code":
|
|
return lambda x: bool(re.fullmatch(r"\d{6,10}", x.strip()))
|
|
|
|
if field == "p_1b_forex_ac_no":
|
|
return lambda x: bool(re.search(r"[A-Z0-9X]{6,20}", x.upper()))
|
|
|
|
if field == "p_1b_ifsc_no":
|
|
return lambda x: bool(re.fullmatch(r"[A-Z]{4}0[A-Z0-9]{6}", x.strip().upper()))
|
|
|
|
if field in {"p_1b_dbk_ac_no", "p_1b_rbi_wr_no"}:
|
|
return lambda x: len(x.strip()) >= 4 and not self._label_is_value(x)
|
|
|
|
if field == "p_1b_rbi_wr_dt":
|
|
return lambda x: self._valid_date(x)
|
|
|
|
if field == "p_1b_type":
|
|
return lambda x: (
|
|
self._norm(x)
|
|
not in {
|
|
"type",
|
|
"gstin type",
|
|
"gstin",
|
|
}
|
|
and not self._label_is_value(x)
|
|
)
|
|
|
|
return lambda x: not self._label_is_value(x)
|
|
|
|
def _extract_party_blocks(self, result):
|
|
"""
|
|
Extract long exporter/consignee blocks using visual labels.
|
|
|
|
The first meaningful line after the anchor is treated as the
|
|
name. Remaining lines in the bounded visual region become the
|
|
address. This is layout based, not sample-value based.
|
|
"""
|
|
|
|
pairs = [
|
|
(
|
|
"p_1b_exporter_name",
|
|
"p_1b_exporter_address",
|
|
[
|
|
"EXPORTER'S NAME & ADDRESS",
|
|
"1.EXPORTER'S NAME & ADDRESS",
|
|
"EXPORTER NAME & ADDRESS",
|
|
],
|
|
[
|
|
"CB NAME",
|
|
"CONSIGNEE NAME & ADDRESS",
|
|
"7.CONSIGNEE NAME & ADDRESS",
|
|
"AD CODE",
|
|
],
|
|
),
|
|
(
|
|
"p_1b_consignee_name",
|
|
"p_1b_consignee_address",
|
|
["CONSIGNEE NAME & ADDRESS"],
|
|
[
|
|
"GSTIN",
|
|
"GSTIN / TYPE",
|
|
"FOREX BANK A/C NO",
|
|
"AD CODE",
|
|
],
|
|
),
|
|
]
|
|
|
|
for name_field, addr_field, starts, stops in pairs:
|
|
label = self._find_label(starts)
|
|
if not label:
|
|
continue
|
|
|
|
lines = self._lines_after_anchor(
|
|
label,
|
|
stop_aliases=stops,
|
|
max_distance=180,
|
|
)
|
|
|
|
lines = [self._clean(x) for x in lines if self._clean(x)]
|
|
lines = [x for x in lines if not self._label_is_value(x)]
|
|
|
|
if not lines:
|
|
continue
|
|
|
|
# Remove obvious unrelated section headings.
|
|
lines = [x for x in lines if not self._is_section_heading(x)]
|
|
|
|
if not lines:
|
|
continue
|
|
|
|
if not result.get(name_field):
|
|
result[name_field] = lines[0]
|
|
self._trace(
|
|
name_field,
|
|
lines[0],
|
|
label["_page"],
|
|
)
|
|
|
|
if not result.get(addr_field):
|
|
address = "\n".join(lines[1:])
|
|
if address:
|
|
result[addr_field] = address
|
|
self._trace(
|
|
addr_field,
|
|
address,
|
|
label["_page"],
|
|
)
|
|
|
|
# =========================================================
|
|
# FINANCIALS
|
|
# =========================================================
|
|
|
|
def _extract_financials(self, result):
|
|
fields = {
|
|
"p_1c_fob_val": ["FOB VALUE"],
|
|
"p_1c_com": ["COMMISSION", "COM"],
|
|
"p_1c_freight": ["FREIGHT"],
|
|
"p_1c_deductions": ["DEDUCTIONS", "DEDUCT"],
|
|
"p_1c_insurance": ["INSURANCE"],
|
|
"p_1c_p_c": ["P & C", "P.C", "P C"],
|
|
"p_1c_discount": ["DISCOUNT"],
|
|
"p_1d_dbk": ["DBK"],
|
|
"p_1d_rodtep_amt": ["RODTEP"],
|
|
"p_1d_rosctle_amt": ["ROSCTL", "ROSL"],
|
|
}
|
|
|
|
for field, aliases in fields.items():
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
continue
|
|
|
|
value = self._value_right_same_row(
|
|
label,
|
|
validator=lambda x: self._float(x) is not None,
|
|
max_distance=180,
|
|
)
|
|
|
|
if value is None:
|
|
value = self._value_below_column(
|
|
label,
|
|
validator=lambda x: self._float(x) is not None,
|
|
max_y_gap=80,
|
|
x_tolerance=120,
|
|
)
|
|
|
|
if value is None:
|
|
continue
|
|
|
|
number = self._float(value)
|
|
if number is None:
|
|
continue
|
|
|
|
result[field] = number
|
|
self._trace(
|
|
field,
|
|
number,
|
|
label["_page"],
|
|
)
|
|
|
|
# =========================================================
|
|
# DATES / TIMES
|
|
# =========================================================
|
|
|
|
def _extract_dates_times(self, result):
|
|
date_specs = {
|
|
"shipping_bill_date": ["SB DATE"],
|
|
"p_1j_subm_dt": ["SUBMISSION"],
|
|
"p_1i_leo_dt": ["LEO DATE"],
|
|
"p_1i_exmn_dt": ["EXAMINATION"],
|
|
"p_1i_brc_realzn_dt": [
|
|
"BRC REALISATION DATE",
|
|
"BRC REALIZATION DATE",
|
|
],
|
|
}
|
|
|
|
for field, aliases in date_specs.items():
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
continue
|
|
|
|
value = self._value_right_same_row(
|
|
label,
|
|
validator=self._valid_date,
|
|
max_distance=220,
|
|
)
|
|
|
|
if value is None:
|
|
value = self._value_below_column(
|
|
label,
|
|
validator=self._valid_date,
|
|
max_y_gap=100,
|
|
x_tolerance=130,
|
|
)
|
|
|
|
if value:
|
|
formatted = self._format_date(value)
|
|
result[field] = formatted
|
|
self._trace(
|
|
field,
|
|
formatted,
|
|
label["_page"],
|
|
)
|
|
|
|
time_specs = {
|
|
"p_1i_subm_time": ["SUBMISSION"],
|
|
"p_1i_exmn_time": ["EXAMINATION"],
|
|
"p_1i_leo_time": ["LEO TIME", "LEO"],
|
|
}
|
|
|
|
for field, aliases in time_specs.items():
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
continue
|
|
|
|
value = self._value_right_same_row(
|
|
label,
|
|
validator=lambda x: bool(
|
|
re.fullmatch(
|
|
r"\d{1,2}:\d{2}(?::\d{2})?",
|
|
x,
|
|
)
|
|
),
|
|
max_distance=220,
|
|
)
|
|
|
|
if value is None:
|
|
value = self._nearest_time(
|
|
label,
|
|
max_y_gap=100,
|
|
)
|
|
|
|
if value:
|
|
formatted = self._format_time(value)
|
|
result[field] = formatted
|
|
self._trace(
|
|
field,
|
|
formatted,
|
|
label["_page"],
|
|
)
|
|
|
|
leo_no = self._find_label(["LEO NO"])
|
|
if leo_no:
|
|
value = self._value_right_same_row(
|
|
leo_no,
|
|
validator=lambda x: bool(re.fullmatch(r"\d+/\d+", x)),
|
|
max_distance=180,
|
|
)
|
|
|
|
if value is None:
|
|
value = self._nearest_token(
|
|
leo_no,
|
|
lambda x: bool(re.fullmatch(r"\d+/\d+", x)),
|
|
max_y_gap=100,
|
|
)
|
|
|
|
if value:
|
|
result["p_1i_leo_no"] = value
|
|
self._trace(
|
|
"p_1i_leo_no",
|
|
value,
|
|
leo_no["_page"],
|
|
)
|
|
|
|
# =========================================================
|
|
# CHILD TABLES
|
|
# =========================================================
|
|
|
|
def _extract_child_tables(self, result):
|
|
"""
|
|
Use pdfplumber's detected tables.
|
|
|
|
No table is assumed to belong to a particular PDF page.
|
|
Classification is based on header content.
|
|
|
|
Unknown tables are intentionally not forced into an
|
|
incorrect child table.
|
|
"""
|
|
|
|
for table_info in self.tables:
|
|
rows = table_info["rows"]
|
|
if not rows:
|
|
continue
|
|
|
|
clean_rows = self._clean_table(rows)
|
|
if not clean_rows:
|
|
continue
|
|
|
|
section_text = self._compact_label(" ".join(self._table_row_text(r) for r in clean_rows[:12]))
|
|
|
|
# Part-II Invoice and Part-III Item tables are handled by the
|
|
# section-aware mappers below. Do not also pass these fragmented
|
|
# tables through the generic classifier, otherwise continuation
|
|
# rows are mistaken for additional child records.
|
|
if "partiiinvoicedetails" in section_text or "partiiiitemdetails" in section_text:
|
|
continue
|
|
|
|
table_name = self._classify_table(clean_rows)
|
|
|
|
if not table_name:
|
|
continue
|
|
|
|
records = self._map_table(
|
|
table_name,
|
|
clean_rows,
|
|
)
|
|
|
|
for record in records:
|
|
result[table_name].append(record)
|
|
|
|
self._trace(
|
|
table_name,
|
|
record,
|
|
table_info["page"],
|
|
)
|
|
|
|
# Merge records that come from fragmented pdfplumber tables.
|
|
# Part III and Part II are commonly split into multiple physical
|
|
# tables, so deduplication alone would lose fields.
|
|
result["item_details"] = self._merge_records_by_key(
|
|
result["item_details"],
|
|
["p_3a_invsno", "p_3a_itemsn"],
|
|
)
|
|
result["item_details"] = self._collapse_item_continuations(result["item_details"])
|
|
result["invoice_details"] = self._merge_invoice_records(result["invoice_details"])
|
|
|
|
for table in self.CHILD_TABLES:
|
|
result[table] = self._deduplicate_records(result[table])
|
|
|
|
# =========================================================
|
|
# TABLE CLASSIFICATION
|
|
# =========================================================
|
|
|
|
def _schema_specific_table_candidate(self, text):
|
|
"""Prefer a child table whose own schema headers are strongly present."""
|
|
candidates = []
|
|
for table_name in self.CHILD_TABLES:
|
|
if table_name in {
|
|
"manifest_details", "annex_details", "invoice_details",
|
|
"item_details", "single_window_declaration", "container_details",
|
|
}:
|
|
continue
|
|
|
|
fields = self.COMPLETE_CHILD_FIELDS.get(table_name, [])
|
|
if not fields:
|
|
continue
|
|
|
|
field_hits = set()
|
|
for field in fields:
|
|
for alias in self._child_field_aliases(table_name, field):
|
|
token = self._compact_header_token(alias)
|
|
if token and token in text:
|
|
field_hits.add(field)
|
|
break
|
|
|
|
table_alias_hits = 0
|
|
for alias in self.TABLE_ALIASES.get(table_name, []):
|
|
token = self._compact_header_token(alias)
|
|
if token and token in text:
|
|
table_alias_hits += 1
|
|
|
|
if len(field_hits) >= 2 and (table_alias_hits >= 1 or len(field_hits) >= 3):
|
|
candidates.append((len(field_hits) * 10 + table_alias_hits, table_name))
|
|
|
|
if not candidates:
|
|
return None
|
|
return max(candidates, key=lambda item: item[0])[1]
|
|
|
|
def _classify_table(self, rows):
|
|
"""Classify tables using section-specific signatures.
|
|
|
|
Generic aliases such as ``invoice`` or ``info`` are intentionally not
|
|
enough to classify a table. This prevents fragmented pdfplumber tables
|
|
from being assigned to the wrong child table.
|
|
"""
|
|
text = re.sub(
|
|
r"[^a-z0-9]+",
|
|
"",
|
|
" ".join(" ".join(str(c or "") for c in row) for row in rows[:8]).lower(),
|
|
)
|
|
|
|
signatures = {
|
|
"manifest_details": (
|
|
["mawbno", "cinno"],
|
|
["mawbdt", "cindt", "cinsiteid"],
|
|
),
|
|
"annex_details": (
|
|
["sealtype", "natureofcargo", "noofpackets"],
|
|
["noofcontainers", "loosepackets", "marksnumbers"],
|
|
),
|
|
"invoice_details": (
|
|
["invoiceno", "invoiceamount"],
|
|
["invterm", "adcode", "invoicevalue", "exchange", "exportername", "buyername"],
|
|
),
|
|
"item_details": (
|
|
["invsn", "itemsn", "hscd", "description"],
|
|
[
|
|
"quantity",
|
|
"uqc",
|
|
"rate",
|
|
"value",
|
|
"dutyamt",
|
|
"igststat",
|
|
"schcod",
|
|
"stateoforigin",
|
|
"districtoforigin",
|
|
"enduse",
|
|
],
|
|
),
|
|
"single_window_declaration": (
|
|
["invsn", "itmsn", "info", "qualifier"],
|
|
["infocd", "infotext", "infomsr", "uqc"],
|
|
),
|
|
}
|
|
|
|
scores = {}
|
|
for table, (required, optional) in signatures.items():
|
|
req_score = sum(1 for token in required if token in text)
|
|
opt_score = sum(1 for token in optional if token in text)
|
|
minimum = 2 if table not in {"annex_details"} else 2
|
|
if req_score >= minimum or opt_score >= 2:
|
|
scores[table] = req_score * 10 + opt_score
|
|
|
|
# Before accepting a broad Single Window signature, prefer a
|
|
# schema-specific child table when its own headers are present.
|
|
# This prevents generic headers such as INVSN/ITMSN/TYPE from
|
|
# incorrectly classifying AA/DFIA, Job, AR4, Supporting Document,
|
|
# Third Party, Manufacturer, RODTEP, etc. as Single Window.
|
|
schema_specific = self._schema_specific_table_candidate(text)
|
|
if schema_specific:
|
|
return schema_specific
|
|
|
|
if scores:
|
|
best = max(scores, key=scores.get)
|
|
# A single-window table can contain only abbreviated data rows
|
|
# (CHR/DTY/ORC) after pdfplumber splits the header away.
|
|
if best != "single_window_declaration":
|
|
if self._looks_like_single_window_rows(rows):
|
|
return "single_window_declaration"
|
|
return best
|
|
|
|
# Headerless/fragmented child tables.
|
|
if self._looks_like_single_window_rows(rows):
|
|
return "single_window_declaration"
|
|
|
|
if self._looks_like_manifest_rows(rows):
|
|
return "manifest_details"
|
|
|
|
if self._looks_like_annex_rows(rows):
|
|
return "annex_details"
|
|
|
|
if self._looks_like_container_rows(rows):
|
|
return "container_details"
|
|
|
|
# Schema-driven classification for every remaining child table.
|
|
# This is intentionally evaluated only after the strict, known-table
|
|
# signatures above so existing invoice/item/single-window behavior is
|
|
# not changed.
|
|
generic_scores = {}
|
|
for table_name in self.CHILD_TABLES:
|
|
if table_name in {
|
|
"manifest_details", "annex_details", "invoice_details",
|
|
"item_details", "single_window_declaration", "container_details",
|
|
}:
|
|
continue
|
|
|
|
aliases = self.TABLE_ALIASES.get(table_name, [])
|
|
fields = self.COMPLETE_CHILD_FIELDS.get(table_name, [])
|
|
all_aliases = list(aliases)
|
|
for field in fields:
|
|
all_aliases.extend(self._child_field_aliases(table_name, field))
|
|
|
|
matched = set()
|
|
for alias in all_aliases:
|
|
token = self._compact_header_token(alias)
|
|
if token and token in text:
|
|
matched.add(token)
|
|
|
|
# Count unique canonical fields represented by the table text.
|
|
field_hits = set()
|
|
for field in fields:
|
|
for alias in self._child_field_aliases(table_name, field):
|
|
token = self._compact_header_token(alias)
|
|
if token and token in text:
|
|
field_hits.add(field)
|
|
break
|
|
|
|
table_alias_hits = sum(
|
|
1 for alias in aliases
|
|
if self._compact_header_token(alias) in text
|
|
)
|
|
if len(field_hits) >= 2 or table_alias_hits >= 2:
|
|
generic_scores[table_name] = len(field_hits) * 10 + table_alias_hits
|
|
|
|
if generic_scores:
|
|
return max(generic_scores, key=generic_scores.get)
|
|
|
|
return None
|
|
|
|
# =========================================================
|
|
# TABLE MAPPING
|
|
# =========================================================
|
|
|
|
def _clean_consignee_address(self, result):
|
|
"""Remove GSTIN/footer bleed from consignee address and recover GSTIN."""
|
|
address = str(result.get("p_1b_consignee_address") or "").strip()
|
|
|
|
if not address:
|
|
return
|
|
|
|
gstin_pattern = re.compile(
|
|
r"\b\d{2}[A-Z]{5}\d{4}[A-Z]\d[A-Z0-9][A-Z0-9]\b",
|
|
re.I,
|
|
)
|
|
|
|
# If GSTIN was swallowed by the address extractor, recover it first.
|
|
address_gstin = gstin_pattern.search(address)
|
|
if address_gstin:
|
|
gstin = address_gstin.group(0).upper()
|
|
|
|
if not result.get("p_1b_gstin"):
|
|
result["p_1b_gstin"] = gstin
|
|
self._trace("p_1b_gstin", gstin)
|
|
|
|
address = address[: address_gstin.start()] + address[address_gstin.end() :]
|
|
|
|
# Remove any remaining GSTIN-like fragment.
|
|
address = gstin_pattern.sub("", address)
|
|
|
|
address = re.sub(
|
|
r"\b(?:GSTIN|GSTN|GSN)\b.*$",
|
|
"",
|
|
address,
|
|
flags=re.I,
|
|
)
|
|
|
|
cleaned_lines = []
|
|
|
|
for raw in address.splitlines():
|
|
line = self._clean(raw)
|
|
|
|
if not line:
|
|
continue
|
|
|
|
if len(line) <= 4 and not re.search(r"\d", line):
|
|
continue
|
|
|
|
if re.fullmatch(
|
|
r"(?:O\s+SA|SA|O|GSN|GSTN|GSTIN)",
|
|
line,
|
|
re.I,
|
|
):
|
|
continue
|
|
|
|
cleaned_lines.append(line)
|
|
|
|
if cleaned_lines:
|
|
result["p_1b_consignee_address"] = "\n".join(cleaned_lines).strip()
|
|
else:
|
|
result["p_1b_consignee_address"] = ""
|
|
|
|
self._trace(
|
|
"p_1b_consignee_address",
|
|
result["p_1b_consignee_address"],
|
|
)
|
|
|
|
def _recover_special_child_tables(self, result):
|
|
"""
|
|
Recover optional child tables without relying on table classification.
|
|
|
|
The Shipping Bill PDF can expose these tables differently from page
|
|
to page. Therefore:
|
|
1. use already mapped valid records when available;
|
|
2. inspect visual rows;
|
|
3. use strict keyword/code patterns;
|
|
4. never fabricate a record from unrelated header values.
|
|
"""
|
|
text = str(self.full_text or "")
|
|
upper = text.upper()
|
|
row_groups = self._group_words_into_rows(self.words)
|
|
|
|
# =========================================================
|
|
# MANIFEST DETAILS
|
|
# =========================================================
|
|
manifest = []
|
|
|
|
# Known/reference Shipping Bill structure:
|
|
# MAWB : 8-12 digit number
|
|
# CIN : long alphanumeric identifier
|
|
# CIN DATE : DD-MMM-YY
|
|
# SITE ID : INBOM4-like code
|
|
mawb_re = re.compile(r"\b\d{8,12}\b")
|
|
cin_re = re.compile(
|
|
r"\b(?=[A-Z0-9]{16,24}\b)"
|
|
r"(?=[A-Z0-9]*[A-Z])"
|
|
r"(?=[A-Z0-9]*\d)"
|
|
r"[A-Z0-9]{16,24}\b",
|
|
re.I,
|
|
)
|
|
date_re = re.compile(
|
|
r"\b\d{1,2}-[A-Z]{3}-\d{2,4}\b",
|
|
re.I,
|
|
)
|
|
site_re = re.compile(
|
|
r"\b[A-Z]{2}[A-Z]{3}\d\b",
|
|
re.I,
|
|
)
|
|
|
|
iec = self._clean(result.get("iec", ""))
|
|
|
|
for index, row in enumerate(row_groups):
|
|
row_text = " ".join(
|
|
self._clean(w.get("_text", "")) for w in row if self._clean(w.get("_text", ""))
|
|
)
|
|
|
|
if not row_text:
|
|
continue
|
|
|
|
# Manifest table header is a strong signal.
|
|
header_signal = re.search(
|
|
r"MAWB|MASTER\s+AWB|CIN\s*(?:NO|DATE|SITE)",
|
|
row_text,
|
|
re.I,
|
|
)
|
|
|
|
# Inspect up to 3 consecutive visual rows.
|
|
block_rows = row_groups[index : index + 3]
|
|
block_text = " ".join(
|
|
" ".join(self._clean(w.get("_text", "")) for w in block if self._clean(w.get("_text", "")))
|
|
for block in block_rows
|
|
)
|
|
|
|
mawbs = [m.group(0) for m in mawb_re.finditer(block_text) if not iec or m.group(0) != iec]
|
|
|
|
cins = [m.group(0).upper() for m in cin_re.finditer(block_text)]
|
|
|
|
dates = [m.group(0).upper() for m in date_re.finditer(block_text)]
|
|
|
|
sites = [m.group(0).upper() for m in site_re.finditer(block_text)]
|
|
|
|
# Reject obvious destination/address strings.
|
|
cins = [
|
|
value
|
|
for value in cins
|
|
if sum(ch.isdigit() for ch in value) >= 4 and sum(ch.isalpha() for ch in value) >= 4
|
|
]
|
|
|
|
sites = [value for value in sites if value.startswith(("IN", "AE", "SA", "US", "GB"))]
|
|
|
|
if mawbs and cins and dates and sites and (header_signal or "MANIFEST" in upper):
|
|
manifest.append(
|
|
{
|
|
"doctype": "Manifest Detail",
|
|
"p_1e_mawb_no": mawbs[0],
|
|
"p_1e_cin_no": cins[0],
|
|
"p_1e_cin_dt": dates[0],
|
|
"p_1e_cin_site_id": sites[0],
|
|
}
|
|
)
|
|
break
|
|
|
|
# Validate existing mapped manifest records.
|
|
for record in result.get("manifest_details", []):
|
|
mawb = self._clean(record.get("p_1e_mawb_no", ""))
|
|
cin = self._clean(record.get("p_1e_cin_no", "")).upper()
|
|
site = self._clean(record.get("p_1e_cin_site_id", "")).upper()
|
|
date = self._clean(record.get("p_1e_cin_dt", ""))
|
|
|
|
if (
|
|
re.fullmatch(r"\d{8,12}", mawb)
|
|
and (not iec or mawb != iec)
|
|
and re.fullmatch(
|
|
r"(?=[A-Z0-9]{16,24}$)"
|
|
r"(?=[A-Z0-9]*[A-Z])"
|
|
r"(?=[A-Z0-9]*\d)"
|
|
r"[A-Z0-9]{16,24}",
|
|
cin,
|
|
re.I,
|
|
)
|
|
and re.fullmatch(
|
|
r"\d{1,2}-[A-Z]{3}-\d{2,4}",
|
|
date,
|
|
re.I,
|
|
)
|
|
and re.fullmatch(
|
|
r"[A-Z]{2}[A-Z]{3}\d",
|
|
site,
|
|
)
|
|
):
|
|
manifest.append(record)
|
|
|
|
result["manifest_details"] = self._deduplicate_records(manifest)
|
|
|
|
# =========================================================
|
|
# ANNEX DETAILS
|
|
# =========================================================
|
|
annex = []
|
|
|
|
# Keep existing valid annex records.
|
|
for record in result.get("annex_details", []):
|
|
if any(
|
|
self._clean(record.get(key, ""))
|
|
for key in (
|
|
"p_1i_seal_typ",
|
|
"p_1i_nature_cargo",
|
|
"p_1i_marks_numbers",
|
|
)
|
|
):
|
|
annex.append(record)
|
|
|
|
# Text fallback for the reference form.
|
|
seal_match = re.search(
|
|
r"\b(WAREHOUSE\s+SEALED|FACTORY\s+SEALED|SELF\s+SEALED)\b",
|
|
text,
|
|
re.I,
|
|
)
|
|
|
|
nature_match = re.search(
|
|
r"\b(PACKAGED|LOOSE)\b",
|
|
text,
|
|
re.I,
|
|
)
|
|
|
|
if not annex and (seal_match or nature_match):
|
|
pkg_value = result.get("pkg", 0) or 0
|
|
cont_value = result.get("cont", 0) or 0
|
|
|
|
# Look for explicit package/container counts if available.
|
|
pkg_match = re.search(
|
|
r"(?:NO\.?\s*OF\s*PKGS|NO\.?\s*OF\s*PACKAGES)"
|
|
r"[^0-9]{0,60}(\d+)",
|
|
text,
|
|
re.I,
|
|
)
|
|
|
|
cont_match = re.search(
|
|
r"(?:NO\.?\s*OF\s*CONTAINERS)"
|
|
r"[^0-9]{0,60}(\d+)",
|
|
text,
|
|
re.I,
|
|
)
|
|
|
|
if pkg_match:
|
|
pkg_value = int(pkg_match.group(1))
|
|
|
|
if cont_match:
|
|
cont_value = int(cont_match.group(1))
|
|
|
|
# Marks/numbers normally follow the corresponding label.
|
|
marks_match = re.search(
|
|
r"MARKS\s*(?:&|AND)\s*NUMBERS?"
|
|
r"\s*[:\-]?\s*(.{20,500})",
|
|
text,
|
|
re.I,
|
|
)
|
|
|
|
annex.append(
|
|
{
|
|
"doctype": "Annex Detail",
|
|
"p_1i_seal_typ": (seal_match.group(1).upper() if seal_match else ""),
|
|
"p_1i_loose_pkts": 0,
|
|
"p_1i_nature_cargo": (nature_match.group(1).upper() if nature_match else ""),
|
|
"p_1i_marks_numbers": (self._clean(marks_match.group(1)) if marks_match else ""),
|
|
"p_1i_no_of_pkgs": pkg_value,
|
|
"p_1i_no_of_containers": cont_value,
|
|
}
|
|
)
|
|
|
|
result["annex_details"] = self._deduplicate_records(annex)
|
|
|
|
# =========================================================
|
|
# SINGLE WINDOW DECLARATION
|
|
# =========================================================
|
|
single = []
|
|
|
|
# Existing records are accepted only when their semantic identity
|
|
# is valid. This removes shifted/duplicate records.
|
|
valid_qualifiers = {"SQC", "GCESS", "RDT", "DOO", "EPT", "STO"}
|
|
valid_info = {"CHR", "DTY", "ORC"}
|
|
|
|
for record in result.get("single_window_declaration", []):
|
|
info = self._clean(record.get("p_4d_info", "")).upper()
|
|
qualifier = self._clean(record.get("p_4d_qualifier", "")).upper()
|
|
|
|
inv = self._clean(record.get("p_4d_invsn", ""))
|
|
item = self._clean(record.get("p_4d_itmsn", ""))
|
|
|
|
if (
|
|
info in valid_info
|
|
and qualifier in valid_qualifiers
|
|
and re.fullmatch(r"\d+", inv)
|
|
and re.fullmatch(r"\d+", item)
|
|
):
|
|
cleaned = {
|
|
"doctype": "Single Window Declaration",
|
|
"p_4d_invsn": inv,
|
|
"p_4d_itmsn": item,
|
|
"p_4d_info": info,
|
|
"p_4d_qualifier": qualifier,
|
|
"p_4d_info_cd": self._clean(record.get("p_4d_info_cd", "")),
|
|
"p_4d_info_text": self._clean(record.get("p_4d_info_text", "")),
|
|
"p_4d_info_msr": self._clean(record.get("p_4d_info_msr", "")),
|
|
"p_4d_uqc": self._clean(record.get("p_4d_uqc", "")).upper(),
|
|
}
|
|
single.append(cleaned)
|
|
|
|
# Visual-row fallback. This does not require extract_tables().
|
|
for row in row_groups:
|
|
cells = [
|
|
self._clean(w.get("_text", ""))
|
|
for w in sorted(
|
|
row,
|
|
key=lambda w: float(w.get("x0", 0)),
|
|
)
|
|
if self._clean(w.get("_text", ""))
|
|
]
|
|
|
|
if not cells:
|
|
continue
|
|
|
|
info_index = next(
|
|
(i for i, value in enumerate(cells) if value.upper() in valid_info),
|
|
None,
|
|
)
|
|
|
|
qual_index = next(
|
|
(i for i, value in enumerate(cells) if value.upper() in valid_qualifiers),
|
|
None,
|
|
)
|
|
|
|
if info_index is None or qual_index is None:
|
|
continue
|
|
|
|
if qual_index <= info_index:
|
|
continue
|
|
|
|
before = [value for value in cells[:info_index] if re.fullmatch(r"\d+", value)]
|
|
|
|
if len(before) < 2:
|
|
continue
|
|
|
|
inv = before[-2]
|
|
item = before[-1]
|
|
info = cells[info_index].upper()
|
|
qualifier = cells[qual_index].upper()
|
|
after = cells[qual_index + 1 :]
|
|
|
|
record = {
|
|
"doctype": "Single Window Declaration",
|
|
"p_4d_invsn": inv,
|
|
"p_4d_itmsn": item,
|
|
"p_4d_info": info,
|
|
"p_4d_qualifier": qualifier,
|
|
"p_4d_info_cd": "",
|
|
"p_4d_info_text": "",
|
|
"p_4d_info_msr": "",
|
|
"p_4d_uqc": "",
|
|
}
|
|
|
|
if qualifier == "SQC":
|
|
for value in after:
|
|
if re.fullmatch(r"\d+(?:\.\d+)?", value):
|
|
record["p_4d_info_msr"] = value
|
|
break
|
|
|
|
for value in after:
|
|
if value.upper() in {"NOS", "KGS", "SET"}:
|
|
record["p_4d_uqc"] = value.upper()
|
|
break
|
|
|
|
elif qualifier == "GCESS":
|
|
for value in after:
|
|
if re.fullmatch(r"\d+(?:\.\d+)?", value):
|
|
record["p_4d_info_msr"] = value
|
|
break
|
|
|
|
for value in after:
|
|
if value.upper() in {"INR", "USD", "EUR", "GBP"}:
|
|
record["p_4d_uqc"] = value.upper()
|
|
break
|
|
|
|
elif qualifier == "RDT":
|
|
if after:
|
|
record["p_4d_info_cd"] = after[0].upper()
|
|
if len(after) > 1:
|
|
record["p_4d_info_text"] = " ".join(after[1:])
|
|
|
|
elif qualifier in {"DOO", "EPT", "STO"}:
|
|
if after:
|
|
record["p_4d_info_cd"] = after[0].upper()
|
|
|
|
single.append(record)
|
|
|
|
# Semantic deduplication.
|
|
unique_single = []
|
|
seen_single = set()
|
|
|
|
for record in single:
|
|
key = (
|
|
record.get("p_4d_invsn", ""),
|
|
record.get("p_4d_itmsn", ""),
|
|
record.get("p_4d_info", ""),
|
|
record.get("p_4d_qualifier", ""),
|
|
)
|
|
|
|
if key in seen_single:
|
|
continue
|
|
|
|
seen_single.add(key)
|
|
unique_single.append(record)
|
|
|
|
result["single_window_declaration"] = unique_single
|
|
|
|
def _looks_like_single_window_rows(self, rows):
|
|
text = " ".join(self._table_row_text(r) for r in rows[:15]).upper()
|
|
codes = len(re.findall(r"\b(?:CHR|DTY|ORC)\b", text))
|
|
qualifiers = len(re.findall(r"\b(?:SQC|GCESS|RDT|DOO|EPT|STO)\b", text))
|
|
return codes >= 1 and qualifiers >= 1
|
|
|
|
def _looks_like_manifest_rows(self, rows):
|
|
text = self._compact_label(" ".join(self._table_row_text(r) for r in rows[:10]))
|
|
return ("mawb" in text and "cin" in text) or ("mawbno" in text and "cinsiteid" in text)
|
|
|
|
def _looks_like_annex_rows(self, rows):
|
|
text = self._compact_label(" ".join(self._table_row_text(r) for r in rows[:10]))
|
|
return ("sealtype" in text and "natureofcargo" in text) or (
|
|
"marksnumbers" in text and "noofcontainers" in text
|
|
)
|
|
|
|
def _looks_like_container_rows(self, rows):
|
|
text = self._compact_label(" ".join(self._table_row_text(r) for r in rows[:10]))
|
|
return "containerno" in text or ("container" in text and "seal" in text)
|
|
|
|
def _map_table(self, table_name, rows):
|
|
if table_name == "invoice_details":
|
|
return self._map_invoice_table(rows)
|
|
|
|
if table_name == "item_details":
|
|
return self._map_item_table(rows)
|
|
|
|
if table_name == "single_window_declaration":
|
|
return self._map_single_window_table(rows)
|
|
|
|
if table_name == "manifest_details":
|
|
return self._map_manifest_table(rows)
|
|
|
|
if table_name == "annex_details":
|
|
return self._map_annex_table(rows)
|
|
|
|
return self._map_generic_table(
|
|
table_name,
|
|
rows,
|
|
)
|
|
|
|
def _map_invoice_table(self, rows):
|
|
header_index = self._header_row(
|
|
rows,
|
|
[
|
|
"invoice",
|
|
"inv",
|
|
"invoice no",
|
|
"inv no",
|
|
],
|
|
)
|
|
|
|
if header_index is None:
|
|
header_index = 0
|
|
|
|
headers = self._make_headers(rows[header_index])
|
|
|
|
records = []
|
|
|
|
for row in rows[header_index + 1 :]:
|
|
if self._empty_row(row):
|
|
continue
|
|
|
|
values = self._row_values(
|
|
headers,
|
|
row,
|
|
)
|
|
|
|
if not any(values.values()):
|
|
continue
|
|
|
|
record = {
|
|
"doctype": "Invoice Detail",
|
|
}
|
|
|
|
self._set_first(
|
|
record,
|
|
"inv_sn",
|
|
values,
|
|
["s no", "s.no", "sn", "serial"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2a_inv_no",
|
|
values,
|
|
["invoice no", "inv no"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2a_inv_dt",
|
|
values,
|
|
["invoice date", "inv dt", "date"],
|
|
date=True,
|
|
)
|
|
|
|
# Many Shipping Bills place invoice number and date in the same
|
|
# cell (for example: ``6031962610 29/07/2026``). Split that
|
|
# combined cell without depending on the sample value.
|
|
if not record.get("p_2a_inv_dt"):
|
|
raw_invoice = record.get("p_2a_inv_no", "")
|
|
date_match = re.search(
|
|
r"\b\d{1,2}[/-]\d{1,2}[/-]\d{2,4}\b",
|
|
raw_invoice,
|
|
)
|
|
if date_match:
|
|
record["p_2a_inv_dt"] = self._format_date(date_match.group(0))
|
|
record["p_2a_inv_no"] = raw_invoice[: date_match.start()].strip()
|
|
self._set_first(
|
|
record,
|
|
"p_2a_ad_code",
|
|
values,
|
|
["ad code"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2a_invterm",
|
|
values,
|
|
["invterm", "term", "incoterm"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_invoice_value",
|
|
values,
|
|
["invoice value", "invoice value fc", "value"],
|
|
number=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_invoice_curr",
|
|
values,
|
|
["currency", "invoice curr"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_fob_val",
|
|
values,
|
|
["fob value", "fob"],
|
|
number=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_fob_curr",
|
|
values,
|
|
["fob curr", "currency"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_freight",
|
|
values,
|
|
["freight"],
|
|
number=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_discount",
|
|
values,
|
|
["discount"],
|
|
number=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_deduct",
|
|
values,
|
|
["deduct", "deduction"],
|
|
number=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_insurance",
|
|
values,
|
|
["insurance"],
|
|
number=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_commison",
|
|
values,
|
|
["commission", "commison", "com"],
|
|
number=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_p_c",
|
|
values,
|
|
["p&c", "p c", "pc"],
|
|
number=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_2c_exchng_rate_desc",
|
|
values,
|
|
["exchange rate", "exchng rate"],
|
|
)
|
|
|
|
record.setdefault(
|
|
"p_2b_exporter_name",
|
|
"",
|
|
)
|
|
record.setdefault(
|
|
"p_2b_exporter_addr",
|
|
"",
|
|
)
|
|
record.setdefault(
|
|
"p_2b_buyer_name",
|
|
"",
|
|
)
|
|
record.setdefault(
|
|
"p_2b_buyer_addr",
|
|
"",
|
|
)
|
|
|
|
records.append(record)
|
|
|
|
return records
|
|
|
|
def _map_item_table(self, rows):
|
|
header_index = self._header_row(
|
|
rows,
|
|
[
|
|
"cth",
|
|
"hs code",
|
|
"item description",
|
|
"description",
|
|
],
|
|
)
|
|
|
|
if header_index is None:
|
|
header_index = 0
|
|
|
|
headers = self._make_headers(rows[header_index])
|
|
|
|
records = []
|
|
|
|
current_record = None
|
|
|
|
for row in rows[header_index + 1 :]:
|
|
if self._empty_row(row):
|
|
continue
|
|
|
|
values = self._row_values(
|
|
headers,
|
|
row,
|
|
)
|
|
|
|
if not any(values.values()):
|
|
continue
|
|
|
|
cth_value = ""
|
|
for key, value in values.items():
|
|
key_norm = self._compact_label(key)
|
|
if any(token in key_norm for token in ("cth", "hscode", "hsn")):
|
|
match = re.search(
|
|
r"(?<!\d)(\d{8})(?!\d)",
|
|
str(value or "").replace(" ", ""),
|
|
)
|
|
if match:
|
|
cth_value = match.group(1)
|
|
break
|
|
|
|
# No valid 8-digit CTH means this is a wrapped continuation,
|
|
# not a new item.
|
|
if not cth_value:
|
|
if current_record is not None:
|
|
fragment = ""
|
|
for key, value in values.items():
|
|
if "description" in self._compact_label(key):
|
|
fragment = self._clean(value)
|
|
break
|
|
if fragment:
|
|
existing = self._clean(str(current_record.get("p_3a_item_desc", "") or ""))
|
|
if fragment not in existing:
|
|
current_record["p_3a_item_desc"] = (existing + " " + fragment).strip()
|
|
continue
|
|
|
|
record = {
|
|
"doctype": "Item Detail",
|
|
"p_3a_cth": cth_value,
|
|
}
|
|
current_record = record
|
|
|
|
self._set_first(
|
|
record,
|
|
"p_3a_invsno",
|
|
values,
|
|
["inv s no", "inv", "invoice"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_itemsn",
|
|
values,
|
|
["item s no", "item", "s no"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_cth",
|
|
values,
|
|
["cth", "hs code", "hsn"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_item_desc",
|
|
values,
|
|
["item description", "description", "desc"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_qty",
|
|
values,
|
|
["qty", "quantity"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_uqc",
|
|
values,
|
|
["uqc", "unit"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_rate",
|
|
values,
|
|
["rate"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_value",
|
|
values,
|
|
["value"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_fob",
|
|
values,
|
|
["fob"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_pmv",
|
|
values,
|
|
["pmv"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_duty_amt",
|
|
values,
|
|
["duty amt", "duty"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_cess_rate",
|
|
values,
|
|
["cess rate"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_cess_amt",
|
|
values,
|
|
["cess amt", "cess"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_dbk_claimed",
|
|
values,
|
|
["dbk claimed", "dbk"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_igststat",
|
|
values,
|
|
["igst stat", "igst status"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_igst_val",
|
|
values,
|
|
["igst value", "igst val"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_igst_amt",
|
|
values,
|
|
["igst amt", "igst amount"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_schcod",
|
|
values,
|
|
["schcod", "scheme code"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_scheme_desc",
|
|
values,
|
|
["scheme desc", "scheme description"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_sqc_mst",
|
|
values,
|
|
["sqc mst", "sqc"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_sqc_uqc",
|
|
values,
|
|
["sqc uqc"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_state_of_origin",
|
|
values,
|
|
["state of origin"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_district_of_origin",
|
|
values,
|
|
["district of origin"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_pt_abroad",
|
|
values,
|
|
["pt abroad"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_comp_cess",
|
|
values,
|
|
["comp cess"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_end_use",
|
|
values,
|
|
["end use"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_benefit_availd",
|
|
values,
|
|
["benefit avail", "benefit"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_reward_benefit",
|
|
values,
|
|
["reward benefit", "reward"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_3a_third_party_item",
|
|
values,
|
|
["third party"],
|
|
)
|
|
|
|
records.append(record)
|
|
|
|
return records
|
|
|
|
def _map_single_window_table(self, rows):
|
|
header_index = self._header_row(
|
|
rows,
|
|
["qualifier", "info", "info code"],
|
|
)
|
|
|
|
# Header may be separated from the data by pdfplumber. In that case
|
|
# parse structural data rows directly.
|
|
if header_index is None:
|
|
records = []
|
|
for row in rows:
|
|
text = self._table_row_text(row)
|
|
if not re.search(r"\b(?:CHR|DTY|ORC)\b", text, re.I):
|
|
continue
|
|
if not re.search(r"\b(?:SQC|GCESS|RDT|DOO|EPT|STO)\b", text, re.I):
|
|
continue
|
|
|
|
cells = [self._clean(x) for x in row if self._clean(x)]
|
|
record = {
|
|
"doctype": "Single Window Declaration",
|
|
"p_4d_invsn": "",
|
|
"p_4d_itmsn": "",
|
|
"p_4d_info": "",
|
|
"p_4d_qualifier": "",
|
|
"p_4d_info_cd": "",
|
|
"p_4d_info_text": "",
|
|
"p_4d_info_msr": "",
|
|
"p_4d_uqc": "",
|
|
}
|
|
|
|
if cells:
|
|
m = re.search(r"\b(\d+)\b", cells[0])
|
|
if m:
|
|
record["p_4d_invsn"] = m.group(1)
|
|
if len(cells) > 1:
|
|
m = re.search(r"\b(\d+)\b", cells[1])
|
|
if m:
|
|
record["p_4d_itmsn"] = m.group(1)
|
|
|
|
m = re.search(r"\b(CHR|DTY|ORC)\b", text, re.I)
|
|
if m:
|
|
record["p_4d_info"] = m.group(1).upper()
|
|
m = re.search(r"\b(SQC|GCESS|RDT|DOO|EPT|STO)\b", text, re.I)
|
|
if m:
|
|
record["p_4d_qualifier"] = m.group(1).upper()
|
|
|
|
# Remove the structural tokens and take remaining cells as
|
|
# code/text/measure/unit in their original order.
|
|
structural = {
|
|
record["p_4d_info"],
|
|
record["p_4d_qualifier"],
|
|
}
|
|
remaining = [
|
|
c
|
|
for c in cells
|
|
if c not in structural
|
|
and c
|
|
not in {
|
|
record["p_4d_invsn"],
|
|
record["p_4d_itmsn"],
|
|
}
|
|
]
|
|
if remaining:
|
|
record["p_4d_info_cd"] = remaining[0]
|
|
if len(remaining) > 1:
|
|
record["p_4d_info_text"] = remaining[1]
|
|
if len(remaining) > 2:
|
|
record["p_4d_info_msr"] = remaining[2]
|
|
if len(remaining) > 3:
|
|
record["p_4d_uqc"] = remaining[3]
|
|
|
|
record = {k: v for k, v in record.items() if v not in ("", None)}
|
|
if record.get("p_4d_info") and record.get("p_4d_qualifier"):
|
|
record.setdefault("doctype", "Single Window Declaration")
|
|
records.append(record)
|
|
return records
|
|
|
|
headers = self._make_headers(rows[header_index])
|
|
records = []
|
|
|
|
for row in rows[header_index + 1 :]:
|
|
if self._empty_row(row):
|
|
continue
|
|
|
|
values = self._row_values(headers, row)
|
|
record = {"doctype": "Single Window Declaration"}
|
|
|
|
self._set_first(record, "p_4d_invsn", values, ["inv sn", "invoice", "inv"])
|
|
self._set_first(record, "p_4d_itmsn", values, ["item sn", "item"])
|
|
self._set_first(record, "p_4d_info", values, ["info"])
|
|
self._set_first(record, "p_4d_qualifier", values, ["qualifier"])
|
|
self._set_first(record, "p_4d_info_cd", values, ["info code", "info cd", "code"])
|
|
self._set_first(record, "p_4d_info_text", values, ["info text", "text"])
|
|
self._set_first(record, "p_4d_info_msr", values, ["info msr", "measure", "measurement"])
|
|
self._set_first(record, "p_4d_uqc", values, ["uqc", "unit"])
|
|
|
|
if any(v not in ("", None) for v in record.values()):
|
|
records.append(record)
|
|
|
|
return records
|
|
|
|
def _map_manifest_table(self, rows):
|
|
headers = self._make_headers(rows[0])
|
|
records = []
|
|
|
|
for row in rows[1:]:
|
|
if self._empty_row(row):
|
|
continue
|
|
|
|
values = self._row_values(
|
|
headers,
|
|
row,
|
|
)
|
|
|
|
record = {
|
|
"doctype": "Manifest Detail",
|
|
}
|
|
|
|
self._set_first(
|
|
record,
|
|
"p_1e_mawb_no",
|
|
values,
|
|
["mawb no", "mawb"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_1e_cin_no",
|
|
values,
|
|
["cin no", "cin"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_1e_cin_dt",
|
|
values,
|
|
["cin dt", "cin date"],
|
|
date=True,
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_1e_cin_site_id",
|
|
values,
|
|
["cin site id", "site id"],
|
|
)
|
|
|
|
records.append(record)
|
|
|
|
return records
|
|
|
|
def _map_annex_table(self, rows):
|
|
headers = self._make_headers(rows[0])
|
|
records = []
|
|
|
|
for row in rows[1:]:
|
|
if self._empty_row(row):
|
|
continue
|
|
|
|
values = self._row_values(
|
|
headers,
|
|
row,
|
|
)
|
|
|
|
record = {
|
|
"doctype": "Annex Detail",
|
|
}
|
|
|
|
self._set_first(
|
|
record,
|
|
"p_1i_seal_typ",
|
|
values,
|
|
["seal type", "seal"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_1i_loose_pkts",
|
|
values,
|
|
["loose pkts", "loose packages", "loose packets"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_1i_nature_cargo",
|
|
values,
|
|
["nature of cargo", "cargo"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_1i_marks_numbers",
|
|
values,
|
|
["marks numbers", "marks", "numbers"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_1i_no_of_pkgs",
|
|
values,
|
|
["no of pkgs", "no of packets", "packages", "packets", "no packages"],
|
|
)
|
|
self._set_first(
|
|
record,
|
|
"p_1i_no_of_containers",
|
|
values,
|
|
["no of containers", "containers", "no containers"],
|
|
)
|
|
|
|
records.append(record)
|
|
|
|
return records
|
|
|
|
def _canonical_child_field_for_header(self, table_name, header):
|
|
"""Return the existing Frappe child field matching a PDF header."""
|
|
fields = self.COMPLETE_CHILD_FIELDS.get(table_name, [])
|
|
if not fields:
|
|
return None
|
|
|
|
header_compact = self._compact_header_token(header)
|
|
if not header_compact:
|
|
return None
|
|
|
|
best = None
|
|
for field in fields:
|
|
aliases = self._child_field_aliases(table_name, field)
|
|
for alias in aliases:
|
|
alias_compact = self._compact_header_token(alias)
|
|
if not alias_compact:
|
|
continue
|
|
|
|
score = 0
|
|
if header_compact == alias_compact:
|
|
score = 1000 + len(alias_compact)
|
|
elif len(alias_compact) >= 5 and alias_compact in header_compact:
|
|
score = 700 + len(alias_compact)
|
|
elif len(header_compact) >= 5 and header_compact in alias_compact:
|
|
score = 600 + len(header_compact)
|
|
else:
|
|
# Token overlap is useful for variations such as
|
|
# "DOCUMENT NUMBER" vs "DOC NO" but must be strong enough
|
|
# to avoid generic matches such as "TYPE".
|
|
h_tokens = set(re.findall(r"[a-z0-9]+", header_compact))
|
|
a_tokens = set(re.findall(r"[a-z0-9]+", alias_compact))
|
|
overlap = len(h_tokens & a_tokens)
|
|
if overlap >= 2:
|
|
score = 100 + overlap * 10 + min(len(alias_compact), 40)
|
|
|
|
if score and (best is None or score > best[0]):
|
|
best = (score, field)
|
|
|
|
return best[1] if best else None
|
|
|
|
def _compact_header_token(self, value):
|
|
value = re.sub(r"^\s*\d+\s*[.)-]\s*", "", str(value or ""))
|
|
value = re.sub(r"\b(?:1|2|3|4|5|6|7|8|9)\s*[.]?", " ", value)
|
|
return re.sub(r"[^a-z0-9]+", "", value.lower())
|
|
|
|
def _child_field_aliases(self, table_name, field):
|
|
aliases = list(self.CHILD_FIELD_ALIASES.get(table_name, {}).get(field, []))
|
|
|
|
# The canonical Frappe fieldname itself is always a valid alias.
|
|
aliases.append(field)
|
|
|
|
# Add the suffix without the technical section prefix:
|
|
# p_4i_cont_seal -> cont seal / container seal variants.
|
|
suffix = re.sub(r"^p_[0-9]+[a-z]?_", "", field)
|
|
aliases.append(suffix.replace("_", " "))
|
|
|
|
# Expand common abbreviations deterministically.
|
|
for alias in list(aliases):
|
|
text = str(alias)
|
|
aliases.extend([
|
|
text.replace(" NO", " NUMBER"),
|
|
text.replace(" NUMBER", " NO"),
|
|
text.replace(" DT", " DATE"),
|
|
text.replace(" DATE", " DT"),
|
|
text.replace(" QTY", " QUANTITY"),
|
|
text.replace(" QUANTITY", " QTY"),
|
|
text.replace(" AMT", " AMOUNT"),
|
|
text.replace(" AMOUNT", " AMT"),
|
|
])
|
|
|
|
# De-duplicate while preserving order.
|
|
seen = set()
|
|
result = []
|
|
for alias in aliases:
|
|
key = self._compact_header_token(alias)
|
|
if not key or key in seen:
|
|
continue
|
|
seen.add(key)
|
|
result.append(alias)
|
|
return result
|
|
|
|
def _canonical_table_header_index(self, table_name, rows):
|
|
"""Find the row containing the strongest canonical child headers."""
|
|
fields = self.COMPLETE_CHILD_FIELDS.get(table_name, [])
|
|
if not fields:
|
|
return 0
|
|
|
|
best_index = None
|
|
best_score = 0
|
|
for index, row in enumerate(rows[:8]):
|
|
score = 0
|
|
used = set()
|
|
for cell in row:
|
|
field = self._canonical_child_field_for_header(table_name, cell)
|
|
if field and field not in used:
|
|
score += 1
|
|
used.add(field)
|
|
if score > best_score:
|
|
best_score = score
|
|
best_index = index
|
|
|
|
return best_index if best_index is not None and best_score >= 1 else 0
|
|
|
|
def _map_generic_table(self, table_name, rows):
|
|
"""
|
|
Schema-driven generic child-table mapper.
|
|
|
|
Every recognized PDF header is mapped to an EXISTING Frappe child
|
|
fieldname from COMPLETE_CHILD_FIELDS. Raw PDF header names are never
|
|
used as output keys. Unknown columns are ignored rather than creating
|
|
invented keys that cannot be inserted into the DocType.
|
|
"""
|
|
if len(rows) < 2:
|
|
return []
|
|
|
|
header_index = self._canonical_table_header_index(table_name, rows)
|
|
headers = self._make_headers(rows[header_index])
|
|
field_by_column = [
|
|
self._canonical_child_field_for_header(table_name, header)
|
|
for header in headers
|
|
]
|
|
|
|
# Require at least one canonical field to be recognized. This prevents
|
|
# page headings/glossary blocks from becoming child rows.
|
|
if not any(field_by_column):
|
|
return []
|
|
|
|
doctype = self.CHILD_DOCTYPES.get(
|
|
table_name,
|
|
self._doctype_for_table(table_name),
|
|
)
|
|
records = []
|
|
|
|
for row in rows[header_index + 1:]:
|
|
if self._empty_row(row):
|
|
continue
|
|
|
|
row_text = self._table_row_text(row)
|
|
if self._is_section_heading(row_text):
|
|
continue
|
|
|
|
record = {"doctype": doctype}
|
|
mapped_count = 0
|
|
|
|
for index, value in enumerate(row):
|
|
value = "" if value is None else str(value).strip()
|
|
if not value or index >= len(field_by_column):
|
|
continue
|
|
|
|
field = field_by_column[index]
|
|
if not field:
|
|
continue
|
|
|
|
# Do not overwrite a populated field from a duplicate header.
|
|
if record.get(field) not in (None, ""):
|
|
continue
|
|
|
|
record[field] = self._normalize_child_value(value, field)
|
|
mapped_count += 1
|
|
|
|
if mapped_count:
|
|
records.append(record)
|
|
|
|
return records
|
|
|
|
def _normalize_child_value(self, value, field):
|
|
"""Normalize a child value without changing the existing field contract."""
|
|
value = self._clean(value)
|
|
if not value:
|
|
return ""
|
|
|
|
# These fields are intentionally kept as strings in the existing JSON
|
|
# contract even though they contain digits.
|
|
if field in {
|
|
"inv_sn", "p_2a_inv_no", "p_2a_ad_code",
|
|
"p_3a_invsno", "p_3a_itemsn", "p_3a_qty", "p_3a_rate",
|
|
"p_3a_value", "p_3a_fob", "p_3a_pmv", "p_3a_sqc_mst",
|
|
"p_4d_invsn", "p_4d_itmsn", "p_4d_info_msr",
|
|
} or field.endswith("_curr"):
|
|
return value
|
|
|
|
if field.endswith("_dt") or field.endswith("_date"):
|
|
return self._format_date(value)
|
|
|
|
if any(token in field.lower() for token in (
|
|
"amount", "value", "rate", "quantity", "qty", "percentage",
|
|
"cess", "duty", "igst", "freight", "discount", "insurance",
|
|
"deduct", "commission", "commison", "pmv", "no_of_units",
|
|
)):
|
|
number = self._float(value)
|
|
if number is not None:
|
|
return number
|
|
|
|
return value
|
|
|
|
"""
|
|
Generic safe mapper.
|
|
|
|
Unknown child-table columns are preserved as deterministic
|
|
key/value pairs instead of inventing semantic mappings.
|
|
"""
|
|
|
|
if len(rows) < 2:
|
|
return []
|
|
|
|
headers = self._make_headers(rows[0])
|
|
records = []
|
|
|
|
doctype = self._doctype_for_table(table_name)
|
|
|
|
for row in rows[1:]:
|
|
if self._empty_row(row):
|
|
continue
|
|
|
|
values = self._row_values(
|
|
headers,
|
|
row,
|
|
)
|
|
|
|
if not any(values.values()):
|
|
continue
|
|
|
|
record = {
|
|
"doctype": doctype,
|
|
}
|
|
|
|
for index, value in enumerate(row):
|
|
if not value:
|
|
continue
|
|
|
|
header = headers[index] if index < len(headers) else f"column_{index + 1}"
|
|
|
|
key = self._safe_key(header)
|
|
|
|
if key:
|
|
record[key] = self._normalize_cell(
|
|
value,
|
|
key,
|
|
)
|
|
|
|
records.append(record)
|
|
|
|
return records
|
|
|
|
# =========================================================
|
|
# FALLBACK EXTRACTION
|
|
# =========================================================
|
|
|
|
def _fallback_text_extraction(self, result):
|
|
"""
|
|
Small fallback layer only for fields where the visual label
|
|
was not found.
|
|
|
|
It is deliberately conservative. It never overrides a value
|
|
already extracted from the visual model.
|
|
"""
|
|
|
|
# ---------------------------------------------------------
|
|
# GROSS WEIGHT / PACKAGES
|
|
# ---------------------------------------------------------
|
|
# Do not depend on exact line breaks. pdfplumber layout=True
|
|
# may produce:
|
|
#
|
|
# G.WT
|
|
# 1143 KGS
|
|
#
|
|
# or:
|
|
#
|
|
# G . W T 1143 KGS
|
|
#
|
|
# or place the value before the label. We therefore search a
|
|
# bounded text window around the actual label.
|
|
|
|
self._recover_gwt_pkg_from_text(result)
|
|
|
|
if not result["p_1b_gstin"]:
|
|
match = re.search(
|
|
r"\b\d{2}[A-Z]{5}\d{4}[A-Z][A-Z0-9]Z[A-Z0-9]\b",
|
|
self.full_text.upper(),
|
|
)
|
|
if match:
|
|
result["p_1b_gstin"] = match.group(0)
|
|
self._trace(
|
|
"p_1b_gstin",
|
|
match.group(0),
|
|
)
|
|
|
|
if not result["p_1b_ad_code"]:
|
|
match = re.search(
|
|
r"\b\d{6,10}\b",
|
|
self.full_text,
|
|
)
|
|
if match:
|
|
value = match.group(0)
|
|
result["p_1b_ad_code"] = value
|
|
self._trace(
|
|
"p_1b_ad_code",
|
|
value,
|
|
)
|
|
|
|
# =========================================================
|
|
# VISUAL ANCHOR HELPERS
|
|
# =========================================================
|
|
|
|
def _find_label(self, aliases):
|
|
for page in range(
|
|
1,
|
|
len(self.pages) + 1,
|
|
):
|
|
page_words = [w for w in self.words if w["_page"] == page]
|
|
|
|
found = self._find_label_in_words(
|
|
page_words,
|
|
aliases,
|
|
)
|
|
|
|
if found:
|
|
return found
|
|
|
|
return None
|
|
|
|
def _find_label_in_words(
|
|
self,
|
|
words,
|
|
aliases,
|
|
):
|
|
"""
|
|
Find an exact visual label.
|
|
|
|
Important: aliases are tokenized BEFORE normalization. The previous
|
|
implementation normalized ``PORT CODE`` to ``portcode`` and then
|
|
tried to compare it with individual PDF words ``Port`` and ``Code``.
|
|
That made the label engine miss real labels and fall back to unrelated
|
|
text.
|
|
"""
|
|
|
|
def compact(value):
|
|
return re.sub(r"[^a-z0-9]+", "", str(value or "").lower())
|
|
|
|
alias_tokens = []
|
|
alias_compact = set()
|
|
for alias in aliases or []:
|
|
alias_text = str(alias)
|
|
tokens = [compact(x) for x in re.findall(r"[A-Za-z0-9]+", alias_text)]
|
|
tokens = [x for x in tokens if x]
|
|
if tokens:
|
|
alias_tokens.append(tokens)
|
|
alias_compact.add("".join(tokens))
|
|
|
|
# Visual order is mandatory. Never use PDF text-flow order here.
|
|
ordered = sorted(
|
|
words,
|
|
key=lambda w: (w["y0"], w["x0"]),
|
|
)
|
|
|
|
for index, word in enumerate(ordered):
|
|
first_raw = compact(word.get("_text"))
|
|
first = re.sub(r"^\d+", "", first_raw)
|
|
|
|
# Single PDF words such as ``G.WT`` or ``12.PORT`` can
|
|
# represent a multi-token logical label.
|
|
if first_raw in alias_compact or first in alias_compact:
|
|
for alias in aliases or []:
|
|
alias_compact_value = compact(alias)
|
|
if alias_compact_value == first_raw or alias_compact_value == first:
|
|
return self._label_object(ordered, index, index)
|
|
|
|
for parts in alias_tokens:
|
|
if not parts:
|
|
continue
|
|
if first != parts[0] and first_raw != parts[0]:
|
|
continue
|
|
matched = [word]
|
|
previous = word
|
|
|
|
for part in parts[1:]:
|
|
found = None
|
|
for candidate in ordered[index + len(matched) : index + len(matched) + 5]:
|
|
if candidate["_page"] != word["_page"]:
|
|
break
|
|
if abs(candidate["y0"] - word["y0"]) > 4:
|
|
break
|
|
if candidate["x0"] < previous["x0"]:
|
|
continue
|
|
if compact(candidate.get("_text")) == part:
|
|
found = candidate
|
|
break
|
|
|
|
if found is None:
|
|
break
|
|
|
|
matched.append(found)
|
|
previous = found
|
|
|
|
if len(matched) == len(parts):
|
|
return self._label_object(
|
|
ordered,
|
|
ordered.index(matched[0]),
|
|
ordered.index(matched[-1]),
|
|
)
|
|
|
|
return None
|
|
|
|
def _label_object(
|
|
self,
|
|
words,
|
|
start,
|
|
end,
|
|
):
|
|
selected = words[start : end + 1]
|
|
|
|
return {
|
|
"_page": selected[0]["_page"],
|
|
"x0": min(w["x0"] for w in selected),
|
|
"x1": max(w["x1"] for w in selected),
|
|
"y0": min(w["y0"] for w in selected),
|
|
"y1": max(w["y1"] for w in selected),
|
|
"text": " ".join(w["_text"] for w in selected),
|
|
}
|
|
|
|
def _value_below_label(
|
|
self,
|
|
label,
|
|
max_y_gap=80,
|
|
validator=None,
|
|
):
|
|
candidates = []
|
|
|
|
for word in self.words:
|
|
if word["_page"] != label["_page"]:
|
|
continue
|
|
|
|
if word["y0"] < label["y1"]:
|
|
continue
|
|
|
|
dy = word["y0"] - label["y1"]
|
|
if dy > max_y_gap:
|
|
continue
|
|
|
|
center = (word["x0"] + word["x1"]) / 2
|
|
|
|
label_center = (label["x0"] + label["x1"]) / 2
|
|
|
|
dx = abs(center - label_center)
|
|
|
|
if dx > 180:
|
|
continue
|
|
|
|
value = word["_text"]
|
|
|
|
if self._label_is_value(value):
|
|
continue
|
|
|
|
if validator and not validator(value):
|
|
continue
|
|
|
|
candidates.append(
|
|
(
|
|
dy + dx * 0.25,
|
|
word,
|
|
)
|
|
)
|
|
|
|
if not candidates:
|
|
return None
|
|
|
|
candidates.sort(key=lambda x: x[0])
|
|
|
|
return candidates[0][1]["_text"]
|
|
|
|
def _value_in_visual_region(
|
|
self,
|
|
label,
|
|
other_labels,
|
|
validator=None,
|
|
max_y_gap=120,
|
|
):
|
|
"""
|
|
Read a visual block below a label while excluding nearby
|
|
unrelated columns.
|
|
|
|
The horizontal region is based on the label's actual width
|
|
and nearby content rather than a fixed document coordinate.
|
|
"""
|
|
|
|
page_words = [w for w in self.words if w["_page"] == label["_page"]]
|
|
|
|
candidates = []
|
|
|
|
for word in page_words:
|
|
if word["y0"] < label["y1"] + 1:
|
|
continue
|
|
|
|
dy = word["y0"] - label["y1"]
|
|
if dy > max_y_gap:
|
|
continue
|
|
|
|
# Prefer the same visual column.
|
|
if word["x1"] < label["x0"] - 30:
|
|
continue
|
|
|
|
# Prevent crossing far-away columns.
|
|
if word["x0"] > label["x1"] + 320:
|
|
continue
|
|
|
|
if self._label_is_value(word["_text"]):
|
|
continue
|
|
|
|
if validator and not validator(word["_text"]):
|
|
continue
|
|
|
|
candidates.append(word)
|
|
|
|
if not candidates:
|
|
return None
|
|
|
|
candidates.sort(
|
|
key=lambda w: (
|
|
w["y0"],
|
|
w["x0"],
|
|
)
|
|
)
|
|
|
|
rows = self._build_rows(candidates)
|
|
|
|
text_rows = []
|
|
|
|
for row in rows:
|
|
text = self._clean(row["text"])
|
|
|
|
if not text:
|
|
continue
|
|
|
|
if self._looks_like_label(text):
|
|
break
|
|
|
|
text_rows.append(text)
|
|
|
|
if len(text_rows) >= 5:
|
|
break
|
|
|
|
if not text_rows:
|
|
return None
|
|
|
|
return "\n".join(text_rows)
|
|
|
|
def _lines_after_anchor(
|
|
self,
|
|
label,
|
|
stop_aliases,
|
|
max_distance=180,
|
|
):
|
|
stop_norm = [self._norm(x) for aliases in stop_aliases for x in aliases]
|
|
|
|
page_rows = [
|
|
row
|
|
for row in self.rows
|
|
if row["_page"] == label["_page"]
|
|
and row["_y"] >= label["y1"]
|
|
and row["_y"] - label["y1"] <= max_distance
|
|
]
|
|
|
|
result = []
|
|
|
|
for row in page_rows:
|
|
text = self._clean(row["text"])
|
|
norm = self._norm(text)
|
|
|
|
if any(x and x in norm for x in stop_norm):
|
|
break
|
|
|
|
# Keep only content overlapping the label's visual area
|
|
# or extending naturally to the right.
|
|
words = row["words"]
|
|
|
|
if not words:
|
|
continue
|
|
|
|
min_x = min(w["x0"] for w in words)
|
|
|
|
if min_x > label["x1"] + 450:
|
|
continue
|
|
|
|
result.append(text)
|
|
|
|
return result
|
|
|
|
# =========================================================
|
|
# NUMERIC / TOKEN HELPERS
|
|
# =========================================================
|
|
|
|
def _extract_by_anchor(
|
|
self,
|
|
result,
|
|
field,
|
|
validator,
|
|
):
|
|
label = self._find_label(self.LABELS.get(field, []))
|
|
|
|
if not label:
|
|
return
|
|
|
|
value = self._value_below_label(
|
|
label,
|
|
max_y_gap=70,
|
|
validator=validator,
|
|
)
|
|
|
|
if value is not None:
|
|
result[field] = value
|
|
self._trace(
|
|
field,
|
|
value,
|
|
label["_page"],
|
|
)
|
|
|
|
def _extract_numeric_anchor(
|
|
self,
|
|
result,
|
|
field,
|
|
aliases,
|
|
integer=False,
|
|
):
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
return
|
|
|
|
value = self._nearest_number(
|
|
label,
|
|
max_y_gap=70,
|
|
)
|
|
|
|
if value is None:
|
|
return
|
|
|
|
if integer:
|
|
value = int(value)
|
|
|
|
result[field] = value
|
|
self._trace(
|
|
field,
|
|
value,
|
|
label["_page"],
|
|
)
|
|
|
|
def _nearest_number(
|
|
self,
|
|
label,
|
|
max_y_gap=100,
|
|
):
|
|
candidates = []
|
|
|
|
for word in self.words:
|
|
if word["_page"] != label["_page"]:
|
|
continue
|
|
|
|
if word["y0"] < label["y1"]:
|
|
continue
|
|
|
|
dy = word["y0"] - label["y1"]
|
|
if dy > max_y_gap:
|
|
continue
|
|
|
|
value = self._float(word["_text"])
|
|
if value is None:
|
|
continue
|
|
|
|
center1 = (label["x0"] + label["x1"]) / 2
|
|
center2 = (word["x0"] + word["x1"]) / 2
|
|
|
|
dx = abs(center1 - center2)
|
|
|
|
if dx > 220:
|
|
continue
|
|
|
|
candidates.append(
|
|
(
|
|
dy + dx * 0.25,
|
|
value,
|
|
)
|
|
)
|
|
|
|
if not candidates:
|
|
return None
|
|
|
|
candidates.sort()
|
|
return candidates[0][1]
|
|
|
|
def _nearest_date(
|
|
self,
|
|
label,
|
|
max_y_gap=100,
|
|
):
|
|
return self._nearest_token(
|
|
label,
|
|
lambda x: self._valid_date(x),
|
|
max_y_gap,
|
|
)
|
|
|
|
def _nearest_time(
|
|
self,
|
|
label,
|
|
max_y_gap=120,
|
|
):
|
|
return self._nearest_token(
|
|
label,
|
|
lambda x: bool(
|
|
re.fullmatch(
|
|
r"\d{1,2}:\d{2}(?::\d{2})?",
|
|
x,
|
|
)
|
|
),
|
|
max_y_gap,
|
|
)
|
|
|
|
def _nearest_token(
|
|
self,
|
|
label,
|
|
validator,
|
|
max_y_gap=100,
|
|
):
|
|
candidates = []
|
|
|
|
for word in self.words:
|
|
if word["_page"] != label["_page"]:
|
|
continue
|
|
|
|
if word["y0"] < label["y1"]:
|
|
continue
|
|
|
|
dy = word["y0"] - label["y1"]
|
|
if dy > max_y_gap:
|
|
continue
|
|
|
|
value = word["_text"].strip()
|
|
|
|
if not validator(value):
|
|
continue
|
|
|
|
center1 = (label["x0"] + label["x1"]) / 2
|
|
center2 = (word["x0"] + word["x1"]) / 2
|
|
|
|
dx = abs(center1 - center2)
|
|
|
|
if dx > 250:
|
|
continue
|
|
|
|
candidates.append(
|
|
(
|
|
dy + dx * 0.25,
|
|
value,
|
|
)
|
|
)
|
|
|
|
if not candidates:
|
|
return None
|
|
|
|
candidates.sort()
|
|
return candidates[0][1]
|
|
|
|
def _near_unit(self, aliases):
|
|
label = self._find_label(aliases)
|
|
if not label:
|
|
return ""
|
|
|
|
allowed = {
|
|
"KGS",
|
|
"KG",
|
|
"NOS",
|
|
"SET",
|
|
"PCS",
|
|
"PKG",
|
|
"MT",
|
|
}
|
|
|
|
value = self._nearest_token(
|
|
label,
|
|
lambda x: x.upper() in allowed,
|
|
70,
|
|
)
|
|
|
|
return value.upper() if value else ""
|
|
|
|
def _find_iec(self):
|
|
label = self._find_label(["IEC/BR", "IEC"])
|
|
|
|
if not label:
|
|
return ""
|
|
|
|
candidates = []
|
|
label_center = (label["x0"] + label["x1"]) / 2
|
|
|
|
for word in self.words:
|
|
if word["_page"] != label["_page"]:
|
|
continue
|
|
|
|
if word["y0"] < label["y1"] - 2:
|
|
continue
|
|
|
|
dy = word["y0"] - label["y1"]
|
|
|
|
if dy > 90:
|
|
continue
|
|
|
|
value = word["_text"].strip()
|
|
|
|
if not re.fullmatch(r"\d{10,12}", value):
|
|
continue
|
|
|
|
word_center = (word["x0"] + word["x1"]) / 2
|
|
dx = abs(word_center - label_center)
|
|
|
|
if dx > 120:
|
|
continue
|
|
|
|
candidates.append(
|
|
(
|
|
dy + dx * 0.35,
|
|
value,
|
|
)
|
|
)
|
|
|
|
if not candidates:
|
|
return ""
|
|
|
|
candidates.sort(key=lambda x: x[0])
|
|
return candidates[0][1][:10]
|
|
|
|
def _find_iec_branch(self, iec):
|
|
for word in self.words:
|
|
value = word["_text"].strip()
|
|
|
|
if re.fullmatch(
|
|
r"\d{10,12}",
|
|
value,
|
|
):
|
|
if value.startswith(iec):
|
|
return value[len(iec) :] or "0"
|
|
|
|
return "0"
|
|
|
|
# =========================================================
|
|
# VALIDATORS
|
|
# =========================================================
|
|
|
|
def _validator_for(self, field):
|
|
if field == "shipping_bill_date":
|
|
return self._valid_date
|
|
|
|
if field == "port_code":
|
|
return self._valid_port
|
|
|
|
if field == "shipping_bill_no":
|
|
return self._valid_sb_number
|
|
|
|
if field in {
|
|
"cb_code",
|
|
"p_1b_ad_code",
|
|
}:
|
|
return self._valid_code
|
|
|
|
return None
|
|
|
|
@staticmethod
|
|
def _valid_port(value):
|
|
value = value.strip().upper()
|
|
return bool(
|
|
re.fullmatch(
|
|
r"[A-Z]{2,5}\d{1,4}",
|
|
value,
|
|
)
|
|
)
|
|
|
|
@staticmethod
|
|
def _valid_sb_number(value):
|
|
value = value.strip()
|
|
return bool(
|
|
re.fullmatch(
|
|
r"\d{6,12}",
|
|
value,
|
|
)
|
|
)
|
|
|
|
@staticmethod
|
|
def _valid_cb_code(value):
|
|
"""Validate Indian Customs Broker code; reject numeric pincodes."""
|
|
value = str(value or "").strip().upper()
|
|
return bool(re.fullmatch(r"[A-Z]{5}\d{4}[A-Z]{3}\d{3}", value))
|
|
|
|
@staticmethod
|
|
def _valid_code(value):
|
|
value = value.strip()
|
|
return bool(
|
|
re.fullmatch(
|
|
r"[A-Z0-9]{4,20}",
|
|
value.upper(),
|
|
)
|
|
)
|
|
|
|
@staticmethod
|
|
def _valid_date(value):
|
|
value = value.strip().upper()
|
|
|
|
patterns = [
|
|
r"\d{1,2}-[A-Z]{3}-\d{2,4}",
|
|
r"\d{1,2}/\d{1,2}/\d{2,4}",
|
|
r"\d{4}-\d{1,2}-\d{1,2}",
|
|
]
|
|
|
|
return any(re.fullmatch(p, value) for p in patterns)
|
|
|
|
@staticmethod
|
|
def _location_validator(value):
|
|
value = value.strip()
|
|
|
|
if not value:
|
|
return False
|
|
|
|
if re.fullmatch(
|
|
r"[YN]",
|
|
value.upper(),
|
|
):
|
|
return False
|
|
|
|
if re.fullmatch(
|
|
r"\d+",
|
|
value,
|
|
):
|
|
return False
|
|
|
|
return True
|
|
|
|
# =========================================================
|
|
# NORMALIZATION
|
|
# =========================================================
|
|
|
|
def _normalize(self, result):
|
|
for field in self.DATE_FIELDS:
|
|
value = result.get(field)
|
|
|
|
if value:
|
|
result[field] = self._format_date(value)
|
|
|
|
for field in self.NUMBER_FIELDS:
|
|
value = result.get(field)
|
|
|
|
if value in ("", None):
|
|
if field in {
|
|
"gwt",
|
|
"p_1c_fob_val",
|
|
"p_1c_com",
|
|
"p_1c_freight",
|
|
"p_1c_deductions",
|
|
"p_1c_insurance",
|
|
"p_1c_p_c",
|
|
"p_1c_discount",
|
|
"p_1d_dbk",
|
|
"p_1d_rodtep_amt",
|
|
"p_1d_rosctle_amt",
|
|
}:
|
|
result[field] = 0.0
|
|
else:
|
|
result[field] = 0
|
|
|
|
for table in self.CHILD_TABLES:
|
|
for record in result.get(table, []):
|
|
for key, value in list(record.items()):
|
|
if value is None:
|
|
continue
|
|
|
|
if key.endswith(("_dt", "_date")):
|
|
if value:
|
|
record[key] = self._format_date(str(value))
|
|
|
|
elif value == "":
|
|
if key == "p_4d_info_msr":
|
|
continue
|
|
if self._looks_numeric_key(key):
|
|
record[key] = 0.0
|
|
|
|
# Clean accidental whitespace.
|
|
for key, value in result.items():
|
|
if isinstance(value, str):
|
|
if key in {
|
|
"port_of_loading",
|
|
"cntry_of_finaldstn",
|
|
"state_of_origin",
|
|
"port_of_finaldstn",
|
|
"port_of_discharge",
|
|
"cntry_of_discharge",
|
|
}:
|
|
result[key] = self._clean_location(value)
|
|
else:
|
|
result[key] = self._clean(value)
|
|
|
|
if result.get("gwt_unit"):
|
|
result["gwt_unit"] = str(result["gwt_unit"]).upper().strip().rstrip(".")
|
|
|
|
# Keep integer count fields as integers.
|
|
for field in {"inv", "item", "pkg", "cont"}:
|
|
value = result.get(field)
|
|
if isinstance(value, float) and value.is_integer():
|
|
result[field] = int(value)
|
|
|
|
# =========================================================
|
|
# TABLE HELPERS
|
|
# =========================================================
|
|
|
|
@staticmethod
|
|
def _clean_table(table):
|
|
result = []
|
|
|
|
for row in table:
|
|
if not row:
|
|
continue
|
|
|
|
clean = [
|
|
""
|
|
if cell is None
|
|
else re.sub(
|
|
r"\s+",
|
|
" ",
|
|
str(cell).replace(
|
|
"\xa0",
|
|
" ",
|
|
),
|
|
).strip()
|
|
for cell in row
|
|
]
|
|
|
|
if any(clean):
|
|
result.append(clean)
|
|
|
|
return result
|
|
|
|
def _header_row(self, rows, aliases):
|
|
normalized = [self._norm(x) for x in aliases]
|
|
|
|
for index, row in enumerate(rows[:5]):
|
|
text = self._norm(" ".join(str(x or "") for x in row))
|
|
|
|
score = sum(1 for alias in normalized if alias and alias in text)
|
|
|
|
if score:
|
|
return index
|
|
|
|
return None
|
|
|
|
def _make_headers(self, row):
|
|
headers = []
|
|
|
|
for index, cell in enumerate(row):
|
|
header = self._norm(str(cell or ""))
|
|
|
|
if not header:
|
|
header = f"column {index + 1}"
|
|
|
|
headers.append(header)
|
|
|
|
return headers
|
|
|
|
def _row_values(self, headers, row):
|
|
values = {}
|
|
|
|
for index, header in enumerate(headers):
|
|
if index >= len(row):
|
|
values[header] = ""
|
|
else:
|
|
values[header] = "" if row[index] is None else str(row[index]).strip()
|
|
|
|
return values
|
|
|
|
def _set_first(
|
|
self,
|
|
record,
|
|
target,
|
|
values,
|
|
aliases,
|
|
number=False,
|
|
date=False,
|
|
):
|
|
def compact(value):
|
|
return re.sub(r"[^a-z0-9]+", "", str(value or "").lower())
|
|
|
|
normalized_aliases = [compact(x) for x in aliases]
|
|
|
|
for header, value in values.items():
|
|
if not value:
|
|
continue
|
|
|
|
header_compact = compact(header)
|
|
header_compact = re.sub(r"^\d+", "", header_compact)
|
|
if not any(
|
|
alias and (alias in header_compact or header_compact in alias) for alias in normalized_aliases
|
|
):
|
|
continue
|
|
|
|
if date:
|
|
value = self._format_date(value)
|
|
elif number:
|
|
number_value = self._float(value)
|
|
value = number_value if number_value is not None else value
|
|
|
|
record[target] = value
|
|
return
|
|
|
|
def _collapse_item_continuations(self, records):
|
|
"""
|
|
Collapse wrapped/duplicated item rows.
|
|
|
|
A genuine Shipping Bill item is anchored by an 8-digit CTH.
|
|
pdfplumber may additionally return:
|
|
- continuation rows without CTH;
|
|
- the same CTH with split invoice/item identifiers;
|
|
- duplicate visual rows containing the same CTH.
|
|
|
|
Those fragments are merged into the active item instead of being
|
|
returned as additional Item Detail records.
|
|
"""
|
|
if not records:
|
|
return []
|
|
|
|
def valid_cth(value):
|
|
text = str(value or "")
|
|
match = re.search(r"(?<!\d)(\d{8})(?!\d)", text)
|
|
return match.group(1) if match else ""
|
|
|
|
def normalize_id(value):
|
|
text = self._clean(str(value or ""))
|
|
# "1 1", "1-1", and "11" are handled as the same numeric
|
|
# identifier where pdfplumber split a cell.
|
|
digits = re.sub(r"\D", "", text)
|
|
return digits or text.upper()
|
|
|
|
def identity(record):
|
|
return (
|
|
normalize_id(record.get("p_3a_invsno")),
|
|
normalize_id(record.get("p_3a_itemsn")),
|
|
)
|
|
|
|
def core_values(record):
|
|
return any(
|
|
str(record.get(field, "")).strip()
|
|
for field in (
|
|
"p_3a_qty",
|
|
"p_3a_uqc",
|
|
"p_3a_rate",
|
|
"p_3a_value",
|
|
"p_3a_fob",
|
|
"p_3a_pmv",
|
|
)
|
|
)
|
|
|
|
def description(record):
|
|
return self._norm(
|
|
re.sub(
|
|
r"[^A-Za-z0-9 ]+",
|
|
" ",
|
|
str(record.get("p_3a_item_desc", "") or ""),
|
|
)
|
|
)
|
|
|
|
def fragment_like(record, active):
|
|
text = description(record)
|
|
active_text = description(active)
|
|
|
|
if not text:
|
|
return True
|
|
|
|
# A short description fragment such as "IN GM GASM" is not
|
|
# a standalone item when its CTH/identity matches the active row.
|
|
if text in active_text or active_text in text:
|
|
return True
|
|
|
|
# Shared beginning/end words are a strong indication of a
|
|
# wrapped duplicate from the same table row.
|
|
words_a = text.split()
|
|
words_b = active_text.split()
|
|
|
|
if len(words_a) <= 6:
|
|
if words_a and words_a[:2] == words_b[:2]:
|
|
return True
|
|
|
|
return False
|
|
|
|
collapsed = []
|
|
active = None
|
|
|
|
for raw in records:
|
|
record = dict(raw)
|
|
cth = valid_cth(record.get("p_3a_cth"))
|
|
|
|
if active is None:
|
|
if not cth:
|
|
continue
|
|
|
|
record["p_3a_cth"] = cth
|
|
record["doctype"] = "Item Detail"
|
|
active = record
|
|
collapsed.append(active)
|
|
continue
|
|
|
|
active_cth = valid_cth(active.get("p_3a_cth"))
|
|
|
|
if not cth:
|
|
self._merge_item_fragment(active, record)
|
|
continue
|
|
|
|
rid = identity(record)
|
|
aid = identity(active)
|
|
|
|
same_identity = rid != ("", "") and aid != ("", "") and rid == aid
|
|
|
|
same_cth = cth == active_cth
|
|
|
|
# Strong duplicate rule:
|
|
# same CTH + same invoice/item identity, even if one of the
|
|
# identifiers was split differently by pdfplumber.
|
|
strong_duplicate = same_cth and (same_identity or rid == ("", "") or aid == ("", ""))
|
|
|
|
# Same CTH + fragmented description is also a duplicate.
|
|
fragment_duplicate = same_cth and fragment_like(record, active)
|
|
|
|
# Same CTH + no meaningful numeric/core fields is a continuation.
|
|
continuation_duplicate = same_cth and not core_values(record)
|
|
|
|
if strong_duplicate or fragment_duplicate or continuation_duplicate:
|
|
self._merge_item_fragment(active, record)
|
|
continue
|
|
|
|
# A new valid CTH is a new genuine item.
|
|
record["p_3a_cth"] = cth
|
|
record["doctype"] = "Item Detail"
|
|
active = record
|
|
collapsed.append(active)
|
|
|
|
# Final safety pass: normalize identifiers and remove any remaining
|
|
# duplicate CTH records with the same invoice/item identity.
|
|
final = []
|
|
seen = set()
|
|
|
|
for record in collapsed:
|
|
cth = valid_cth(record.get("p_3a_cth"))
|
|
if not cth:
|
|
continue
|
|
|
|
rid = identity(record)
|
|
key = (cth, rid)
|
|
|
|
if key in seen:
|
|
for existing in final:
|
|
if valid_cth(existing.get("p_3a_cth")) == cth and identity(existing) == rid:
|
|
self._merge_item_fragment(existing, record)
|
|
break
|
|
continue
|
|
|
|
seen.add(key)
|
|
record["p_3a_cth"] = cth
|
|
record["doctype"] = "Item Detail"
|
|
final.append(record)
|
|
|
|
return final
|
|
|
|
def _merge_item_fragment(self, target, fragment):
|
|
"""Merge continuation data without overwriting existing values."""
|
|
description = self._clean(str(fragment.get("p_3a_item_desc", "") or ""))
|
|
existing = self._clean(str(target.get("p_3a_item_desc", "") or ""))
|
|
|
|
if description and description not in existing:
|
|
target["p_3a_item_desc"] = (existing + " " + description).strip()
|
|
|
|
for field, value in fragment.items():
|
|
if field in {"doctype", "p_3a_cth", "p_3a_item_desc"}:
|
|
continue
|
|
if value in ("", None):
|
|
continue
|
|
if target.get(field) in ("", None):
|
|
target[field] = value
|
|
|
|
def _merge_records_by_key(self, records, key_fields):
|
|
merged = {}
|
|
loose = []
|
|
|
|
for record in records:
|
|
key = tuple(str(record.get(field, "")).strip() for field in key_fields)
|
|
if not any(key):
|
|
loose.append(record)
|
|
continue
|
|
|
|
if key not in merged:
|
|
merged[key] = dict(record)
|
|
continue
|
|
|
|
for field, value in record.items():
|
|
if value in ("", None, 0, 0.0):
|
|
continue
|
|
if merged[key].get(field) in ("", None, 0, 0.0):
|
|
merged[key][field] = value
|
|
|
|
return list(merged.values()) + loose
|
|
|
|
def _merge_invoice_records(self, records):
|
|
# Prefer invoice number as the correlation key. For fragmented
|
|
# tables that contain no invoice number, inv_sn is used.
|
|
indexed = {}
|
|
loose = []
|
|
|
|
for record in records:
|
|
inv_no = str(record.get("p_2a_inv_no", "")).strip()
|
|
inv_sn = str(record.get("inv_sn", "")).strip()
|
|
key = inv_no or inv_sn
|
|
if not key:
|
|
loose.append(record)
|
|
continue
|
|
|
|
if key not in indexed:
|
|
indexed[key] = dict(record)
|
|
continue
|
|
|
|
for field, value in record.items():
|
|
if value in ("", None, 0, 0.0):
|
|
continue
|
|
if indexed[key].get(field) in ("", None, 0, 0.0):
|
|
indexed[key][field] = value
|
|
|
|
if loose and len(indexed) == 1:
|
|
target = next(iter(indexed.values()))
|
|
for record in loose:
|
|
for field, value in record.items():
|
|
if field == "doctype" or value in ("", None, 0, 0.0):
|
|
continue
|
|
if target.get(field) in ("", None, 0, 0.0):
|
|
target[field] = value
|
|
return list(indexed.values())
|
|
|
|
return list(indexed.values()) + loose
|
|
|
|
def _deduplicate_records(self, records):
|
|
seen = set()
|
|
result = []
|
|
|
|
for record in records:
|
|
key = repr(sorted(record.items()))
|
|
|
|
if key in seen:
|
|
continue
|
|
|
|
seen.add(key)
|
|
result.append(record)
|
|
|
|
return result
|
|
|
|
# =========================================================
|
|
# TEXT / VALUE CLEANING
|
|
# =========================================================
|
|
|
|
@staticmethod
|
|
def _norm(value):
|
|
value = str(value or "").upper()
|
|
value = value.replace("&", " AND ")
|
|
value = re.sub(
|
|
r"[^A-Z0-9]+",
|
|
" ",
|
|
value,
|
|
)
|
|
return (
|
|
re.sub(
|
|
r"\s+",
|
|
" ",
|
|
value,
|
|
)
|
|
.strip()
|
|
.lower()
|
|
)
|
|
|
|
@staticmethod
|
|
def _clean(value):
|
|
if value is None:
|
|
return ""
|
|
|
|
value = str(value)
|
|
value = value.replace(
|
|
"\xa0",
|
|
" ",
|
|
)
|
|
value = re.sub(
|
|
r"[ ]+",
|
|
" ",
|
|
value,
|
|
)
|
|
|
|
return value.strip()
|
|
|
|
@staticmethod
|
|
def _repair_country_text(value):
|
|
"""Repair conservative leading-character OCR/layout noise in countries."""
|
|
value = ShippingBillParser._clean(value)
|
|
if not value:
|
|
return ""
|
|
|
|
# A merged label/value boundary can prepend one stray N. Keep the
|
|
# correction narrow so NIGERIA, NAMIBIA, etc. remain untouched.
|
|
known_multiword_countries = {
|
|
"SAUDI ARABIA", "UNITED ARAB EMIRATES", "UNITED STATES",
|
|
"UNITED KINGDOM", "SOUTH AFRICA", "SOUTH KOREA", "NORTH KOREA",
|
|
"NEW ZEALAND", "SRI LANKA", "COSTA RICA", "CZECH REPUBLIC",
|
|
"DOMINICAN REPUBLIC", "EL SALVADOR", "PAPUA NEW GUINEA",
|
|
"TRINIDAD AND TOBAGO", "BOSNIA AND HERZEGOVINA", "SAINT KITTS AND NEVIS",
|
|
"SAINT LUCIA", "SAINT VINCENT AND THE GRENADINES", "ANTIGUA AND BARBUDA",
|
|
"EQUATORIAL GUINEA", "BURKINA FASO", "SIERRA LEONE", "SOUTH SUDAN",
|
|
"WESTERN SAHARA", "NORTH MACEDONIA", "MARSHALL ISLANDS", "SOLOMON ISLANDS",
|
|
"CAPE VERDE", "IVORY COAST", "TIMOR LESTE", "BRUNEI DARUSSALAM",
|
|
"VATICAN CITY", "PALESTINE STATE",
|
|
}
|
|
|
|
upper = value.upper()
|
|
for country in known_multiword_countries:
|
|
if upper == "N" + country:
|
|
return value[1:].lstrip()
|
|
|
|
return value
|
|
|
|
def _clean_location(self, value):
|
|
"""
|
|
Clean location values without destroying legitimate word spacing.
|
|
|
|
pdfplumber layout extraction can produce micro-spaces inside a word,
|
|
e.g. "M aharashtra". Only a capital-letter + whitespace + lowercase
|
|
word pattern is joined; normal uppercase multi-word locations such as
|
|
"SAUDI ARABIA" remain unchanged.
|
|
"""
|
|
lines = []
|
|
|
|
for line in str(value).splitlines():
|
|
line = self._clean(line)
|
|
|
|
if not line:
|
|
continue
|
|
|
|
line = re.sub(
|
|
r"(?<=[A-Z])\s+(?=[a-z]{2,})",
|
|
"",
|
|
line,
|
|
)
|
|
|
|
# Also repair repeated one-letter splits such as:
|
|
# "M aharashtra" -> "Maharashtra"
|
|
line = re.sub(
|
|
r"\b([A-Z])\s+([a-z]{3,})\b",
|
|
r"\1\2",
|
|
line,
|
|
)
|
|
|
|
# Remove obvious section labels accidentally adjacent to
|
|
# the actual value.
|
|
line = re.sub(
|
|
r"^\d+\.\s*(PORT|COUNTRY|STATE)\s+[^ ]+\s*",
|
|
"",
|
|
line,
|
|
flags=re.I,
|
|
).strip()
|
|
|
|
if line:
|
|
lines.append(line)
|
|
|
|
return "\n".join(self._repair_country_text(x) for x in lines)
|
|
|
|
def _clean_party_value(self, field, value):
|
|
value = self._clean(value)
|
|
|
|
if field == "p_1b_gstin":
|
|
match = re.search(
|
|
r"\b\d{2}[A-Z]{5}\d{4}[A-Z][A-Z0-9]Z[A-Z0-9]\b",
|
|
value.upper(),
|
|
)
|
|
if match:
|
|
return match.group(0)
|
|
|
|
if field == "p_1b_ad_code":
|
|
match = re.search(
|
|
r"\b\d{6,10}\b",
|
|
value,
|
|
)
|
|
if match:
|
|
return match.group(0)
|
|
|
|
if field in {"p_1b_exporter_name", "p_1b_consignee_name"}:
|
|
value = re.sub(r"^(?:P|O|SA)\s+(?=[A-Z])", "", value, flags=re.I).strip()
|
|
|
|
if field == "p_1b_forex_ac_no":
|
|
match = re.search(
|
|
r"\b[A-Z0-9X]{6,20}\b",
|
|
value.upper(),
|
|
)
|
|
if match:
|
|
return match.group(0)
|
|
|
|
return value
|
|
|
|
@staticmethod
|
|
def _looks_like_label(value):
|
|
norm = re.sub(
|
|
r"\s+",
|
|
" ",
|
|
str(value or "").upper(),
|
|
).strip()
|
|
|
|
return bool(
|
|
re.match(
|
|
r"^\d+\.\s*[A-Z][A-Z0-9 /&.'-]{3,}$",
|
|
norm,
|
|
)
|
|
)
|
|
|
|
@staticmethod
|
|
def _is_section_heading(value):
|
|
norm = value.upper()
|
|
|
|
return any(
|
|
x in norm
|
|
for x in [
|
|
"PART -",
|
|
"PART I",
|
|
"PART II",
|
|
"PART III",
|
|
"PART IV",
|
|
"DECLARATION",
|
|
"DIGITALLY SIGNED",
|
|
"SCAN QR",
|
|
]
|
|
)
|
|
|
|
@staticmethod
|
|
def _safe_key(value):
|
|
value = (
|
|
re.sub(
|
|
r"[^a-zA-Z0-9]+",
|
|
"_",
|
|
value,
|
|
)
|
|
.strip("_")
|
|
.lower()
|
|
)
|
|
|
|
if not value:
|
|
return ""
|
|
|
|
if value[0].isdigit():
|
|
value = "column_" + value
|
|
|
|
return value
|
|
|
|
@staticmethod
|
|
def _normalize_cell(value, key):
|
|
value = str(value).strip()
|
|
|
|
if key.endswith(("_dt", "_date")):
|
|
return ShippingBillParser._format_date(value)
|
|
|
|
if ShippingBillParser._looks_numeric_key(key):
|
|
number = ShippingBillParser._float(value)
|
|
if number is not None:
|
|
return number
|
|
|
|
return value
|
|
|
|
@staticmethod
|
|
def _looks_numeric_key(key):
|
|
key = key.lower()
|
|
|
|
return any(
|
|
x in key
|
|
for x in [
|
|
"amt",
|
|
"amount",
|
|
"value",
|
|
"val",
|
|
"qty",
|
|
"quantity",
|
|
"rate",
|
|
"fob",
|
|
"pmv",
|
|
"cess",
|
|
"freight",
|
|
"discount",
|
|
"insurance",
|
|
"deduct",
|
|
"commission",
|
|
"com",
|
|
"pkg",
|
|
"container",
|
|
"measure",
|
|
]
|
|
)
|
|
|
|
@staticmethod
|
|
def _empty_row(row):
|
|
return not any(str(x or "").strip() for x in row)
|
|
|
|
@staticmethod
|
|
def _doctype_for_table(table_name):
|
|
words = table_name.split("_")
|
|
|
|
return " ".join(word.capitalize() for word in words).replace(
|
|
"Details",
|
|
"Detail",
|
|
)
|
|
|
|
# =========================================================
|
|
# NUMBER / DATE UTILITIES
|
|
# =========================================================
|
|
|
|
@staticmethod
|
|
def _num(value):
|
|
try:
|
|
return float(value or 0)
|
|
except (TypeError, ValueError):
|
|
return 0.0
|
|
|
|
@staticmethod
|
|
def _integer(value):
|
|
if value is None:
|
|
return None
|
|
|
|
value = (
|
|
str(value)
|
|
.replace(
|
|
",",
|
|
"",
|
|
)
|
|
.strip()
|
|
)
|
|
|
|
match = re.fullmatch(
|
|
r"\d+",
|
|
value,
|
|
)
|
|
|
|
if not match:
|
|
return None
|
|
|
|
return int(value)
|
|
|
|
@staticmethod
|
|
def _float(value):
|
|
if value is None:
|
|
return None
|
|
|
|
value = str(value)
|
|
value = value.replace(
|
|
",",
|
|
"",
|
|
).strip()
|
|
|
|
match = re.fullmatch(
|
|
r"-?\d+(?:\.\d+)?",
|
|
value,
|
|
)
|
|
|
|
if not match:
|
|
return None
|
|
|
|
try:
|
|
return float(value)
|
|
except ValueError:
|
|
return None
|
|
|
|
@staticmethod
|
|
def _format_date(value):
|
|
if not value:
|
|
return ""
|
|
|
|
value = str(value).strip().upper()
|
|
|
|
# Remove trailing punctuation.
|
|
value = value.strip(".,;:")
|
|
|
|
formats = [
|
|
"%d-%b-%y",
|
|
"%d-%b-%Y",
|
|
"%d/%m/%Y",
|
|
"%d/%m/%y",
|
|
"%d-%m-%Y",
|
|
"%d-%m-%y",
|
|
"%Y-%m-%d",
|
|
"%d.%m.%Y",
|
|
]
|
|
|
|
for fmt in formats:
|
|
try:
|
|
return datetime.strptime(
|
|
value,
|
|
fmt,
|
|
).strftime("%Y-%m-%d")
|
|
except ValueError:
|
|
continue
|
|
|
|
return value
|
|
|
|
@staticmethod
|
|
def _format_time(value):
|
|
if not value:
|
|
return ""
|
|
|
|
match = re.search(
|
|
r"\b(\d{1,2}):(\d{2})(?::(\d{2}))?\b",
|
|
str(value),
|
|
)
|
|
|
|
if not match:
|
|
return str(value).strip()
|
|
|
|
hour, minute, second = match.groups()
|
|
|
|
return f"{int(hour):02d}:{minute}:{second or '00'}"
|
|
|
|
# =========================================================
|
|
# TRACE
|
|
# =========================================================
|
|
|
|
def _trace(
|
|
self,
|
|
field,
|
|
value,
|
|
page=None,
|
|
):
|
|
entry = {
|
|
"field": field,
|
|
"value": value,
|
|
}
|
|
|
|
if page is not None:
|
|
entry["page"] = page
|
|
|
|
self.trace.append(entry)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
print("ShippingBillParser (pdfplumber, deterministic, non-AI) loaded successfully")
|
|
|