bhardwaj08sarthak commited on
Commit
21d8b09
·
verified ·
1 Parent(s): 956e59c

Upload folder using huggingface_hub

Browse files
Files changed (4) hide show
  1. __init__.py +0 -0
  2. docs.py +169 -0
  3. sheets.py +147 -0
  4. utils.py +68 -0
__init__.py ADDED
File without changes
docs.py ADDED
@@ -0,0 +1,169 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import zipfile
2
+ import xml.etree.ElementTree as ET
3
+ from pathlib import Path
4
+
5
+ from converter.utils import are_keys_sequential, getMarkedFields, toSnakeCase
6
+
7
+ W = "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}"
8
+
9
+ headerReplacements = [
10
+ ["next_message_id", "next"],
11
+ ["option_next", "options_next"],
12
+ ["option_next_id", "options_next"],
13
+ ["options_next_id", "options_next"]
14
+ ]
15
+
16
+ avoidTransformMarkers = [{
17
+ "identifier": "pixabowl/posts",
18
+ "fields": ["id"]
19
+ }]
20
+
21
+ splitMarkers = [{
22
+ "identifier": "sidetrails/connections/items",
23
+ "fields": ["bio"]
24
+ }]
25
+
26
+
27
+ # --- local .docx -> Google Docs API body.content bridge ---
28
+
29
+ def _para_text(p: ET.Element) -> str:
30
+ parts = []
31
+ for node in p.iter():
32
+ if node.tag == f"{W}t":
33
+ parts.append(node.text or '')
34
+ elif node.tag == f"{W}br":
35
+ parts.append('\n')
36
+ return ''.join(parts)
37
+
38
+ def _cell_content(tc: ET.Element) -> list:
39
+ return [
40
+ {'paragraph': {'elements': [{'textRun': {'content': _para_text(p)}}]}}
41
+ for p in tc.findall(f"{W}p")
42
+ ]
43
+
44
+ def _docx_to_body_content(path: Path) -> list:
45
+ with zipfile.ZipFile(path) as z:
46
+ root = ET.fromstring(z.read("word/document.xml"))
47
+ body = root.find(f"{W}body")
48
+ content = []
49
+ for child in body:
50
+ if child.tag == f"{W}p":
51
+ text = _para_text(child)
52
+ content.append({'paragraph': {'elements': [{'textRun': {'content': text}}]}})
53
+ elif child.tag == f"{W}tbl":
54
+ rows = [
55
+ {'tableCells': [{'content': _cell_content(tc)} for tc in tr.findall(f"{W}tc")]}
56
+ for tr in child.findall(f"{W}tr")
57
+ ]
58
+ content.append({'table': {'tableRows': rows}})
59
+ return content
60
+
61
+
62
+ # --- semantic data extraction (unchanged from owner's docs.py) ---
63
+
64
+ def read_paragraph_element(element):
65
+ text_run = element.get('textRun')
66
+ if not text_run:
67
+ return ''
68
+ return text_run.get('content')
69
+
70
+ def readCell(elements):
71
+ text = ''
72
+ for value in elements:
73
+ if 'paragraph' in value:
74
+ para_elements = value.get('paragraph').get('elements')
75
+ for elem in para_elements:
76
+ text += read_paragraph_element(elem)
77
+ return text
78
+
79
+ def getArrayFields(elements):
80
+ arrayFields = []
81
+ for value in elements:
82
+ if 'table' in value:
83
+ header = []
84
+ table = value.get('table')
85
+ multi = False
86
+ for rowIndex, row in enumerate(table.get('tableRows')):
87
+ cells = row.get('tableCells')
88
+ for cellIndex, cell in enumerate(cells):
89
+ cellContent = readCell(cell.get('content')).rstrip()
90
+ if rowIndex == 0:
91
+ cellContent = toSnakeCase(cellContent)
92
+ for replacement in headerReplacements:
93
+ cellContent = cellContent.replace(replacement[0], replacement[1])
94
+ header.append(cellContent)
95
+ elif cellIndex == 0:
96
+ multi = toSnakeCase(cellContent) == ''
97
+ elif multi:
98
+ if cellContent != '' and header[cellIndex] not in arrayFields:
99
+ arrayFields.append(header[cellIndex])
100
+ return arrayFields
101
+
102
+ def readDocElements(elements, splitFields, avoidTransformFields):
103
+ fileData = []
104
+ arrayFields = getArrayFields(elements)
105
+
106
+ for value in elements:
107
+ if 'table' in value:
108
+ tableData = {}
109
+ header = []
110
+ table = value.get('table')
111
+ rowData = {}
112
+ rowKey = ''
113
+ multi = False
114
+ for rowIndex, row in enumerate(table.get('tableRows')):
115
+ newRowKey = ''
116
+ cells = row.get('tableCells')
117
+ for cellIndex, cell in enumerate(cells):
118
+ cellContent = readCell(cell.get('content')).rstrip()
119
+ if rowIndex == 0:
120
+ cellContent = toSnakeCase(cellContent)
121
+ for replacement in headerReplacements:
122
+ cellContent = cellContent.replace(replacement[0], replacement[1])
123
+ header.append(cellContent)
124
+ elif cellIndex == 0:
125
+ if avoidTransformFields is not None and header[cellIndex] in avoidTransformFields:
126
+ newRowKey = cellContent
127
+ else:
128
+ newRowKey = toSnakeCase(cellContent)
129
+ if newRowKey != '':
130
+ multi = False
131
+ rowKey = newRowKey
132
+ rowData = {}
133
+ else:
134
+ multi = True
135
+ else:
136
+ if (multi and cellContent == '') or header[cellIndex] == '':
137
+ continue
138
+ cellContent = cellContent if cellContent != '-' else None
139
+ if header[cellIndex] in arrayFields:
140
+ rowData.setdefault(header[cellIndex], [])
141
+ if cellContent is not None:
142
+ rowData[header[cellIndex]].append(cellContent)
143
+ else:
144
+ if splitFields is not None and header[cellIndex] in splitFields:
145
+ if cellContent != '':
146
+ rowData[header[cellIndex]] = cellContent.splitlines()
147
+ else:
148
+ rowData[header[cellIndex]] = cellContent
149
+
150
+ if rowIndex != 0 and rowKey != '':
151
+ non_empty_cols = len(list(h for h in header if h != ''))
152
+ tableData[rowKey] = rowData if non_empty_cols > 2 else (list(rowData.values())[0] if rowData else None)
153
+ fileData.append(tableData)
154
+
155
+ if len(fileData) == 1:
156
+ fileData = fileData[0]
157
+ if are_keys_sequential(fileData):
158
+ fileData = list(fileData.values())
159
+
160
+ return fileData
161
+
162
+
163
+ def readLocalDoc(path: Path, file_path: str):
164
+ body_content = _docx_to_body_content(path)
165
+ return readDocElements(
166
+ body_content,
167
+ getMarkedFields(file_path, splitMarkers),
168
+ getMarkedFields(file_path, avoidTransformMarkers),
169
+ )
sheets.py ADDED
@@ -0,0 +1,147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import re
2
+ import zipfile
3
+ import xml.etree.ElementTree as ET
4
+ from pathlib import Path
5
+
6
+ from converter.utils import are_keys_sequential, toSnakeCase
7
+ from converter.docs import headerReplacements
8
+
9
+ S = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
10
+ R = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}"
11
+
12
+
13
+ def _col_num(letters: str) -> int:
14
+ n = 0
15
+ for ch in letters:
16
+ n = n * 26 + (ord(ch.upper()) - 64)
17
+ return n - 1
18
+
19
+ def _as_number(text: str) -> int | float:
20
+ f = float(text)
21
+ return int(f) if f.is_integer() else f
22
+
23
+ def _shared_strings(z: zipfile.ZipFile) -> list:
24
+ if "xl/sharedStrings.xml" not in z.namelist():
25
+ return []
26
+ root = ET.fromstring(z.read("xl/sharedStrings.xml"))
27
+ return ["".join(t.text or "" for t in si.iter(f"{S}t"))
28
+ for si in root.findall(f"{S}si")]
29
+
30
+ def _worksheet_path(z: zipfile.ZipFile) -> str:
31
+ wb = ET.fromstring(z.read("xl/workbook.xml"))
32
+ sheet = next(iter(wb.iter(f"{S}sheet")))
33
+ rels = ET.fromstring(z.read("xl/_rels/workbook.xml.rels"))
34
+ relmap = {r.get("Id"): r.get("Target") for r in rels}
35
+ target = relmap.get(sheet.get(f"{R}id"), "worksheets/sheet1.xml")
36
+ path = target[1:] if target.startswith("/") else "xl/" + target
37
+ if path not in z.namelist():
38
+ path = sorted(n for n in z.namelist()
39
+ if n.startswith("xl/worksheets/") and n.endswith(".xml"))[0]
40
+ return path
41
+
42
+ def _cell_value(c: ET.Element, shared: list) -> str | int | float:
43
+ t = c.get("t")
44
+ if t == "s":
45
+ v = c.find(f"{S}v")
46
+ return shared[int(v.text)] if v is not None else ""
47
+ if t == "inlineStr":
48
+ return "".join(x.text or "" for x in c.iter(f"{S}t"))
49
+ if t in ("str", "b"):
50
+ v = c.find(f"{S}v")
51
+ return v.text if v is not None and v.text is not None else ""
52
+ v = c.find(f"{S}v")
53
+ if v is None or v.text is None:
54
+ return ""
55
+ return _as_number(v.text)
56
+
57
+
58
+ def readLocalSheet(path: Path, file_path: str):
59
+ with zipfile.ZipFile(path) as z:
60
+ shared = _shared_strings(z)
61
+ ws_path = _worksheet_path(z)
62
+ ws = ET.fromstring(z.read(ws_path))
63
+
64
+ grid: dict = {}
65
+ max_row = 0
66
+ for c in ws.iter(f"{S}c"):
67
+ ref = c.get("r")
68
+ if not ref:
69
+ continue
70
+ m = re.match(r"([A-Za-z]+)(\d+)", ref)
71
+ col, row = _col_num(m.group(1)), int(m.group(2))
72
+ val = _cell_value(c, shared)
73
+ if val != "":
74
+ grid[(row, col)] = val
75
+ max_row = max(max_row, row)
76
+
77
+ header_width = max((col for (row, col) in grid if row == 1), default=-1) + 1
78
+ raw_header = [str(grid.get((1, k), "")) for k in range(header_width)]
79
+ while raw_header and raw_header[-1] == "":
80
+ raw_header.pop()
81
+ ncols = len(raw_header)
82
+
83
+ header = []
84
+ for h in raw_header:
85
+ h = toSnakeCase(h)
86
+ for replacement in headerReplacements:
87
+ h = h.replace(replacement[0], replacement[1])
88
+ header.append(h)
89
+
90
+ last_row = 1
91
+ for (row, col), val in grid.items():
92
+ if row >= 2 and col < ncols and val != "":
93
+ last_row = max(last_row, row)
94
+
95
+ rows = [
96
+ {header[k]: grid.get((r, k), "") for k in range(ncols)}
97
+ for r in range(2, last_row + 1)
98
+ ]
99
+
100
+ # detect array fields: columns that have values in continuation rows (empty first col)
101
+ array_fields = []
102
+ for row in rows:
103
+ first_val = str(row.get(header[0], ""))
104
+ if toSnakeCase(first_val) == "":
105
+ for field in header[1:]:
106
+ val = row.get(field, "")
107
+ if val != "" and field not in array_fields:
108
+ array_fields.append(field)
109
+
110
+ tableData = {}
111
+ rowKey = ""
112
+ rowData: dict = {}
113
+ for row in rows:
114
+ first_val = str(row.get(header[0], ""))
115
+ new_key = toSnakeCase(first_val)
116
+
117
+ if new_key != "":
118
+ rowKey = new_key
119
+ rowData = {}
120
+ for field in header[1:]:
121
+ if field == "":
122
+ continue
123
+ val = row.get(field, "")
124
+ cell = str(val) if isinstance(val, (int, float)) else val
125
+ cell = cell if cell != "-" else None
126
+ if field in array_fields:
127
+ rowData[field] = [cell] if cell is not None else []
128
+ else:
129
+ rowData[field] = cell
130
+ elif rowKey != "":
131
+ for field in header[1:]:
132
+ if field == "":
133
+ continue
134
+ val = row.get(field, "")
135
+ if val == "":
136
+ continue
137
+ cell = str(val) if isinstance(val, (int, float)) else val
138
+ cell = cell if cell != "-" else None
139
+ if field in array_fields and cell is not None:
140
+ rowData.setdefault(field, []).append(cell)
141
+
142
+ if rowKey != "":
143
+ tableData[rowKey] = rowData if len([h for h in header if h != ""]) > 2 else (list(rowData.values())[0] if rowData else None)
144
+
145
+ if are_keys_sequential(tableData):
146
+ return list(tableData.values())
147
+ return tableData
utils.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import json
2
+ import os
3
+ from re import sub
4
+
5
+ entityNameReplacements = [
6
+ ['adam_s', 'adams'],
7
+ ['zoey_s', 'zoeys'],
8
+ ['adams_phone', 'myphone'],
9
+ ['zoeys_phone', 'herphone'],
10
+ ['adams\'phone', 'myphone'],
11
+ ['zoeys\'phone', 'herphone'],
12
+ ['adamsphone', 'myphone'],
13
+ ['zoeysphone', 'herphone'],
14
+ ['prologue', 'episode0'],
15
+ ['flashback', 'fb'],
16
+ ['cutscenesandsequences', 'special'],
17
+ ['sequences', 'events'],
18
+ ['dad', 'zoeysdad'],
19
+ ['mom', 'zoeysmom'],
20
+ ['png', ''],
21
+ ['jpg', ''],
22
+ ['jpeg', ''],
23
+ ]
24
+
25
+ skipMarkers = [
26
+ '(Not to be translated)',
27
+ 'Media',
28
+ 'Voiceovers and Subtitles',
29
+ 'Store'
30
+ ]
31
+
32
+ def toSimpleCase(s, joiner=''):
33
+ return joiner.join(
34
+ sub('([A-Z][a-z]+)', r' \1',
35
+ sub('([A-Z]+)', r' \1',
36
+ s.replace('-', ' ').replace('\'', '').replace('.', ''))).split()).lower()
37
+
38
+ def toSnakeCase(s):
39
+ return toSimpleCase(s, '_')
40
+
41
+ def toSafeEntityName(s):
42
+ s = toSimpleCase(s)
43
+ for replacement in entityNameReplacements:
44
+ s = s.replace(replacement[0], replacement[1])
45
+ return s
46
+
47
+ def are_keys_sequential(d):
48
+ keys = list(d.keys())
49
+ cursor = None
50
+ for index, key in enumerate(keys):
51
+ if index == 0:
52
+ if key != "0" and key != "1":
53
+ return False
54
+ if not key.isnumeric():
55
+ return False
56
+ if index > 0:
57
+ if int(key) - cursor != 1:
58
+ return False
59
+ cursor = int(key)
60
+ return True
61
+
62
+ def getMarkedFields(filePath, markers):
63
+ fileCheck = [m for m in markers if m["identifier"] in filePath]
64
+ file = None if len(fileCheck) == 0 else fileCheck[0]
65
+ return None if file is None else file["fields"]
66
+
67
+ def is_skippable(name):
68
+ return any(marker in name for marker in skipMarkers)