Fix csv for TAG. (#4454)

### What problem does this PR solve?


### Type of change

- [x] Bug Fix (non-breaking change which fixes an issue)
This commit is contained in:
Kevin Hu 2025-01-13 12:03:18 +08:00 committed by GitHub
parent ecdb2a88bd
commit e098fcf6ad
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194

View File

@ -91,14 +91,14 @@ def chunk(filename, binary=None, lang="Chinese", callback=None, **kwargs):
callback(0.1, "Start to parse.")
txt = get_text(filename, binary)
lines = txt.split("\n")
delimiter = "\t" if any("\t" in line for line in lines) else ","
fails = []
content = ""
res = []
reader = csv.reader(lines, delimiter=delimiter)
reader = csv.reader(lines)
for i, row in enumerate(reader):
row = [r.strip() for r in row if r.strip()]
if len(row) != 2:
content += "\n" + lines[i]
elif len(row) == 2: