Python 如何使用分隔符删除 csv 上的白色-space
Python how to use split delimiter to remove white-space on csv
我正在编写一段代码,将 html 表转换为 csv 文件。我无法弄清楚如何使用字符串拆分删除我正在打印到终端的信息之间的 white-space 。我最好的结果是终端在信息之间打印出很大的差距,这使得导航变得困难。任何信息将不胜感激。
import csv
from bs4 import BeautifulSoup
from termcolor import cprint
html = open("recallist.html").read()
soup = BeautifulSoup(html)
table = soup.find_all('div', {'id': 'PrintArea'})
output_rows = []
recals = 'recallist.csv'
cprint('READING TABLES', 'green')
for table_row in table:
columns = table_row.findAll('td')
output_row = []
for column in columns:
output_row.append(column.text)
output_rows.append(output_row)
with open('recallist.csv', 'w', newline='') as csvfile:
writer = csv.writer(csvfile)
writer.writerows(output_rows)
with open(recals, 'r') as f:
contents = f.read()
for item in contents.split("Date,Customer,Phone,Cell Phone,Removal,Notes"):
for refine in item.split('",,'):
print(refine)
下面列出的 CSV 示例:
Location,,,Date,Customer,Phone,Cell Phone,Removal,Notes,�,�,�,,04/29/19 | 03:00 PM,[9999] FIRST LAST,999-999-9999***,999-999-9999,,"
",,"
","
7.92
",,04/29/19 | 03:30 PM,[123456] FIRST LAST,999-999-9999***,999-999-9999,04/13/2020,"
",,"
","
[=13=].02
",,04/29/19 | 04:00 PM,[123456] FIRST LAST,999-999-9999***,,09/10/2019,"
",,"
","
(2.10)
",,04/29/19 | 04:15 PM,[123456] FIRST LAST,999-999-9999***,,01/09/2020,"
",,"
","
.23
",,04/29/19 | 04:30 PM,[123456] FIRST LAST,999-999-9999***,,05/30/2019,"
",,"
","
[=13=].24
",,04/29/19 | 05:00 PM,[123456] FIRST LAST,999-999-9999***,,07/26/2019,"
",,"
","
([=13=].30)
",,04/29/19 | 07:00 PM,[123456] FIRST LAST,999-999-9999***,999-999-9999,11/15/2019,"
",,"
","
[=13=].06
",,04/29/19 | 07:30 PM,[123456] FIRST LAST,999-999-9999***,,12/12/2019,"
",,"
","
我要实现的格式:
04/29/19 | 03:00 PM,[9999] FIRST LAST,999-999-9999***,999-999-9999,7.92
04/29/19 | 03:30 PM,[99999] FIRST LAST,999-999-9999***,999-999-9999,[=14=].02
ETC.
html 样本以备不时之需:
<tbody><tr class="alt">
<td colspan="5" align="left" style="background-color:668cd9;">Location</td>
<td colspan="5" align="left" style="background-color:668cd9;"></td>
</tr>
<tr align="left" class="GrayBLOCK">
<td></td>
<td>Date</td>
<td>Customer</td>
<td>Phone</td>
<td>Cell Phone</td>
<td>Removal</td>
<td>Notes</td>
<td> </td>
<td> </td>
<td> </td>
</tr>
<tr class="alt">
<td></td>
<td>04/29/19 | 03:00 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[9999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td>999-999-9999</td>
<td></td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
7.92
</td>
</tr>
<tr>
<td></td>
<td>04/29/19 | 03:30 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td>999-999-9999</td>
<td>04/13/2020</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
[=15=].02
</td>
</tr>
<tr class="alt">
<td></td>
<td>04/29/19 | 04:00 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td></td>
<td>09/10/2019</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
(2.10)
</td>
</tr>
<tr>
<td></td>
<td>04/29/19 | 04:15 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td></td>
<td>01/09/2020</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
.23
</td>
</tr>
<tr class="alt">
<td></td>
<td>04/29/19 | 04:30 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td></td>
<td>05/30/2019</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
[=15=].24
</td>
</tr>
<tr>
<td></td>
<td>04/29/19 | 05:00 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td></td>
<td>07/26/2019</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
([=15=].30)
</td>
</tr>
<tr class="alt">
<td></td>
<td>04/29/19 | 07:00 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td>999-999-9999</td>
<td>11/15/2019</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
[=15=].06
</td>
</tr>
更新:我在原来的 post 中发现了一个问题,这里有更好的版本。空的 <td>
标签会创建一些额外的列。版本 1 保留了这些列,版本 2 删除了它们,但它非常特定于您提供的格式,如果格式更改,则必须修改切片。
版本 1
import csv
from bs4 import BeautifulSoup
with open("recallist.html") as f:
soup = BeautifulSoup(f.read(), features="html.parser")
rows = soup.find_all('tr')
with open('recallist.csv', 'w', newline='') as csvfile:
writer = csv.writer(csvfile)
for row in rows:
columns = row.find_all('td')
writer.writerow([column.get_text(strip=True) for column in columns])
版本 2
import csv
from bs4 import BeautifulSoup
with open("recallist.html") as f:
soup = BeautifulSoup(f.read(), features="html.parser")
rows = soup.find_all('tr')
with open('recallist.csv', 'w', newline='') as csvfile:
writer = csv.writer(csvfile)
#alt: 'for row in rows[2:]:' to slice off the two header rows
for row in rows:
columns = row.find_all('td')
del columns[0]
del columns[-4:-1]
writer.writerow([column.get_text(strip=True) for column in columns])
如果您的真实 HTML 实际上有多个包含各种列的表,则需要对此进行调整。希望对您有所帮助!
我正在编写一段代码,将 html 表转换为 csv 文件。我无法弄清楚如何使用字符串拆分删除我正在打印到终端的信息之间的 white-space 。我最好的结果是终端在信息之间打印出很大的差距,这使得导航变得困难。任何信息将不胜感激。
import csv
from bs4 import BeautifulSoup
from termcolor import cprint
html = open("recallist.html").read()
soup = BeautifulSoup(html)
table = soup.find_all('div', {'id': 'PrintArea'})
output_rows = []
recals = 'recallist.csv'
cprint('READING TABLES', 'green')
for table_row in table:
columns = table_row.findAll('td')
output_row = []
for column in columns:
output_row.append(column.text)
output_rows.append(output_row)
with open('recallist.csv', 'w', newline='') as csvfile:
writer = csv.writer(csvfile)
writer.writerows(output_rows)
with open(recals, 'r') as f:
contents = f.read()
for item in contents.split("Date,Customer,Phone,Cell Phone,Removal,Notes"):
for refine in item.split('",,'):
print(refine)
下面列出的 CSV 示例:
Location,,,Date,Customer,Phone,Cell Phone,Removal,Notes,�,�,�,,04/29/19 | 03:00 PM,[9999] FIRST LAST,999-999-9999***,999-999-9999,,"
",,"
","
7.92
",,04/29/19 | 03:30 PM,[123456] FIRST LAST,999-999-9999***,999-999-9999,04/13/2020,"
",,"
","
[=13=].02
",,04/29/19 | 04:00 PM,[123456] FIRST LAST,999-999-9999***,,09/10/2019,"
",,"
","
(2.10)
",,04/29/19 | 04:15 PM,[123456] FIRST LAST,999-999-9999***,,01/09/2020,"
",,"
","
.23
",,04/29/19 | 04:30 PM,[123456] FIRST LAST,999-999-9999***,,05/30/2019,"
",,"
","
[=13=].24
",,04/29/19 | 05:00 PM,[123456] FIRST LAST,999-999-9999***,,07/26/2019,"
",,"
","
([=13=].30)
",,04/29/19 | 07:00 PM,[123456] FIRST LAST,999-999-9999***,999-999-9999,11/15/2019,"
",,"
","
[=13=].06
",,04/29/19 | 07:30 PM,[123456] FIRST LAST,999-999-9999***,,12/12/2019,"
",,"
","
我要实现的格式:
04/29/19 | 03:00 PM,[9999] FIRST LAST,999-999-9999***,999-999-9999,7.92
04/29/19 | 03:30 PM,[99999] FIRST LAST,999-999-9999***,999-999-9999,[=14=].02
ETC.
html 样本以备不时之需:
<tbody><tr class="alt">
<td colspan="5" align="left" style="background-color:668cd9;">Location</td>
<td colspan="5" align="left" style="background-color:668cd9;"></td>
</tr>
<tr align="left" class="GrayBLOCK">
<td></td>
<td>Date</td>
<td>Customer</td>
<td>Phone</td>
<td>Cell Phone</td>
<td>Removal</td>
<td>Notes</td>
<td> </td>
<td> </td>
<td> </td>
</tr>
<tr class="alt">
<td></td>
<td>04/29/19 | 03:00 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[9999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td>999-999-9999</td>
<td></td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
7.92
</td>
</tr>
<tr>
<td></td>
<td>04/29/19 | 03:30 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td>999-999-9999</td>
<td>04/13/2020</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
[=15=].02
</td>
</tr>
<tr class="alt">
<td></td>
<td>04/29/19 | 04:00 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td></td>
<td>09/10/2019</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
(2.10)
</td>
</tr>
<tr>
<td></td>
<td>04/29/19 | 04:15 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td></td>
<td>01/09/2020</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
.23
</td>
</tr>
<tr class="alt">
<td></td>
<td>04/29/19 | 04:30 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td></td>
<td>05/30/2019</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
[=15=].24
</td>
</tr>
<tr>
<td></td>
<td>04/29/19 | 05:00 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td></td>
<td>07/26/2019</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
([=15=].30)
</td>
</tr>
<tr class="alt">
<td></td>
<td>04/29/19 | 07:00 PM</td>
<td><a href="../code/c_newClient.cfm?theID=99999" target="_blank">[999999]</a> FIRST LAST</td>
<td>999-999-9999***</td>
<td>999-999-9999</td>
<td>11/15/2019</td>
<td>
</td>
<td></td>
<td>
</td>
<td align="right" class="RedMED">
[=15=].06
</td>
</tr>
更新:我在原来的 post 中发现了一个问题,这里有更好的版本。空的 <td>
标签会创建一些额外的列。版本 1 保留了这些列,版本 2 删除了它们,但它非常特定于您提供的格式,如果格式更改,则必须修改切片。
版本 1
import csv
from bs4 import BeautifulSoup
with open("recallist.html") as f:
soup = BeautifulSoup(f.read(), features="html.parser")
rows = soup.find_all('tr')
with open('recallist.csv', 'w', newline='') as csvfile:
writer = csv.writer(csvfile)
for row in rows:
columns = row.find_all('td')
writer.writerow([column.get_text(strip=True) for column in columns])
版本 2
import csv
from bs4 import BeautifulSoup
with open("recallist.html") as f:
soup = BeautifulSoup(f.read(), features="html.parser")
rows = soup.find_all('tr')
with open('recallist.csv', 'w', newline='') as csvfile:
writer = csv.writer(csvfile)
#alt: 'for row in rows[2:]:' to slice off the two header rows
for row in rows:
columns = row.find_all('td')
del columns[0]
del columns[-4:-1]
writer.writerow([column.get_text(strip=True) for column in columns])
如果您的真实 HTML 实际上有多个包含各种列的表,则需要对此进行调整。希望对您有所帮助!