mirror of
https://codeberg.org/scip/kleingebaeck.git
synced 2025-12-17 04:21:00 +01:00
Compare commits
2 Commits
ad-conditi
...
v0.3.15
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
09948a6b39 | ||
|
|
bc01391872 |
@@ -204,6 +204,7 @@ Price: 99 € VB
|
||||
Id: 1919191919
|
||||
Category: Sachbücher
|
||||
Condition: Sehr Gut
|
||||
Type: Buch
|
||||
Created: 10.12.2023
|
||||
|
||||
This is the description text.
|
||||
|
||||
12
ad.go
12
ad.go
@@ -31,7 +31,10 @@ type Ad struct {
|
||||
Title string `goquery:"h1"`
|
||||
Slug string
|
||||
ID string
|
||||
Condition string `goquery:".addetailslist--detail--value,text"`
|
||||
Details []string `goquery:".addetailslist--detail--value,text"`
|
||||
Condition string // post processed from details
|
||||
Type string // post processed from details
|
||||
Color string // post processed from details
|
||||
Category string
|
||||
CategoryTree []string `goquery:".breadcrump-link,text"`
|
||||
Price string `goquery:"h2#viewad-price"`
|
||||
@@ -56,6 +59,13 @@ func (ad *Ad) LogValue() slog.Value {
|
||||
)
|
||||
}
|
||||
|
||||
// static set of conditions available, used for post processing details
|
||||
var CONDITIONS = []string{"Neu", "Gut", "Sehr Gut", "In Ordnung"}
|
||||
var COLORS = []string{"Beige", "Blau", "Braun", "Bunt", "Burgunderrot",
|
||||
"Creme", "Gelb", "Gold", "Grau", "Grün", "Holz", "Khaki", "Lavelndel",
|
||||
"Lila", "Orange", "Pink", "Print", "Rot", "Schwarz", "Silber",
|
||||
"Transparent", "Türkis", "Weiß", "Sonstige"}
|
||||
|
||||
// check for completeness. I erected these fields to be mandatory
|
||||
// (though I really don't know if they really are). I consider images
|
||||
// and meta optional. So, if either of the checked fields here is
|
||||
|
||||
@@ -34,17 +34,17 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
VERSION string = "0.3.13"
|
||||
VERSION string = "0.3.15"
|
||||
Baseuri string = "https://www.kleinanzeigen.de"
|
||||
Listuri string = "/s-bestandsliste.html"
|
||||
Defaultdir string = "."
|
||||
|
||||
DefaultTemplate string = "Title: {{.Title}}\nPrice: {{.Price}}\nId: {{.ID}}\n" +
|
||||
"Category: {{.Category}}\nCondition: {{.Condition}}\n" +
|
||||
"Category: {{.Category}}\nCondition: {{.Condition}}\nType: {{.Type}}\nColor: {{.Color}}\n" +
|
||||
"Created: {{.Created}}\nExpire: {{.Expire}}\n\n{{.Text}}\n"
|
||||
|
||||
DefaultTemplateWin string = "Title: {{.Title}}\r\nPrice: {{.Price}}\r\nId: {{.ID}}\r\n" +
|
||||
"Category: {{.Category}}\r\nCondition: {{.Condition}}\r\n" +
|
||||
"Category: {{.Category}}\r\nCondition: {{.Condition}}\r\nType: {{.Type}}\r\nColor: {{.Color}}\r\n" +
|
||||
"Created: {{.Created}}\r\nExpires: {{.Expire}}\r\n\r\n{{.Text}}\r\n"
|
||||
|
||||
DefaultUserAgent string = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " +
|
||||
|
||||
@@ -23,6 +23,7 @@ outdir = "test"
|
||||
#Id: {{.Id}}
|
||||
#Category: {{.Category}}
|
||||
#Condition: {{.Condition}}
|
||||
#Type: {{.Type}}
|
||||
#Created: {{.Created}}
|
||||
|
||||
#{{.Text}}
|
||||
|
||||
1
go.mod
1
go.mod
@@ -23,6 +23,7 @@ require (
|
||||
|
||||
require (
|
||||
github.com/PuerkitoBio/goquery v1.5.1 // indirect
|
||||
github.com/alecthomas/repr v0.4.0 // indirect
|
||||
github.com/andybalholm/cascadia v1.1.0 // indirect
|
||||
github.com/fatih/color v1.16.0 // indirect
|
||||
github.com/fsnotify/fsnotify v1.7.0 // indirect
|
||||
|
||||
2
go.sum
2
go.sum
@@ -3,6 +3,8 @@ astuart.co/goq v1.0.0/go.mod h1:+fokcnFrO8Pw2fj8drdStJvzoMFebJH69rw8IC21rno=
|
||||
github.com/PuerkitoBio/goquery v1.5.0/go.mod h1:qD2PgZ9lccMbQlc7eEOjaeRlFQON7xY8kdmcsrnKqMg=
|
||||
github.com/PuerkitoBio/goquery v1.5.1 h1:PSPBGne8NIUWw+/7vFBV+kG2J/5MOjbzc7154OaKCSE=
|
||||
github.com/PuerkitoBio/goquery v1.5.1/go.mod h1:GsLWisAFVj4WgDibEWF4pvYnkVQBpKBKeU+7zCJoLcc=
|
||||
github.com/alecthomas/repr v0.4.0 h1:GhI2A8MACjfegCPVq9f1FLvIBS+DrQ2KQBFZP1iFzXc=
|
||||
github.com/alecthomas/repr v0.4.0/go.mod h1:Fr0507jx4eOXV7AlPV6AVZLYrLIuIeSOWtW57eE/O/4=
|
||||
github.com/andybalholm/cascadia v1.0.0/go.mod h1:GsXiBklL0woXo1j/WYWtSYYC4ouU9PqHO0sqidkEA4Y=
|
||||
github.com/andybalholm/cascadia v1.1.0 h1:BuuO6sSfQNFRu1LppgbD25Hr2vLYW25JvxHs5zzsLTo=
|
||||
github.com/andybalholm/cascadia v1.1.0/go.mod h1:GsXiBklL0woXo1j/WYWtSYYC4ouU9PqHO0sqidkEA4Y=
|
||||
|
||||
@@ -133,7 +133,7 @@
|
||||
.\" ========================================================================
|
||||
.\"
|
||||
.IX Title "KLEINGEBAECK 1"
|
||||
.TH KLEINGEBAECK 1 "2024-02-10" "1" "User Commands"
|
||||
.TH KLEINGEBAECK 1 "2025-02-06" "1" "User Commands"
|
||||
.\" For nroff, turn off justification. Always turn off hyphenation; it makes
|
||||
.\" way too many mistakes in technical documents.
|
||||
.if n .ad l
|
||||
@@ -174,7 +174,7 @@ well. We use \s-1TOML\s0 as our configuration language. See
|
||||
.PP
|
||||
Format is pretty simple:
|
||||
.PP
|
||||
.Vb 11
|
||||
.Vb 10
|
||||
\& user = 1010101
|
||||
\& loglevel = verbose
|
||||
\& outdir = "test"
|
||||
@@ -185,6 +185,8 @@ Format is pretty simple:
|
||||
\& Id: {{.ID}}
|
||||
\& Category: {{.Category}}
|
||||
\& Condition: {{.Condition}}
|
||||
\& Type: {{.Type}}
|
||||
\& Color: {{.Color}}
|
||||
\& Created: {{.Created}}
|
||||
\&
|
||||
\& {{.Text}}
|
||||
@@ -267,12 +269,13 @@ variables as the ad name template above.
|
||||
.PP
|
||||
This is the default template:
|
||||
.PP
|
||||
.Vb 7
|
||||
.Vb 8
|
||||
\& Title: {{.Title}}
|
||||
\& Price: {{.Price}}
|
||||
\& Id: {{.ID}}
|
||||
\& Category: {{.Category}}
|
||||
\& Condition: {{.Condition}}
|
||||
\& Type: {{.Type}}
|
||||
\& Created: {{.Created}}
|
||||
\& Expire: {{.Expire}}
|
||||
\&
|
||||
|
||||
@@ -46,6 +46,8 @@ CONFIGURATION
|
||||
Id: {{.ID}}
|
||||
Category: {{.Category}}
|
||||
Condition: {{.Condition}}
|
||||
Type: {{.Type}}
|
||||
Color: {{.Color}}
|
||||
Created: {{.Created}}
|
||||
|
||||
{{.Text}}
|
||||
@@ -111,6 +113,7 @@ TEMPLATES
|
||||
Id: {{.ID}}
|
||||
Category: {{.Category}}
|
||||
Condition: {{.Condition}}
|
||||
Type: {{.Type}}
|
||||
Created: {{.Created}}
|
||||
Expire: {{.Expire}}
|
||||
|
||||
|
||||
@@ -46,6 +46,8 @@ Format is pretty simple:
|
||||
Id: {{.ID}}
|
||||
Category: {{.Category}}
|
||||
Condition: {{.Condition}}
|
||||
Type: {{.Type}}
|
||||
Color: {{.Color}}
|
||||
Created: {{.Created}}
|
||||
|
||||
{{.Text}}
|
||||
@@ -131,6 +133,7 @@ This is the default template:
|
||||
Id: {{.ID}}
|
||||
Category: {{.Category}}
|
||||
Condition: {{.Condition}}
|
||||
Type: {{.Type}}
|
||||
Created: {{.Created}}
|
||||
Expire: {{.Expire}}
|
||||
|
||||
|
||||
23
main_test.go
23
main_test.go
@@ -93,6 +93,10 @@ const ADTPL string = `DOCTYPE html>
|
||||
<li class="addetailslist--detail">
|
||||
Zustand<span class="addetailslist--detail--value" >
|
||||
{{ .Condition }}</span>
|
||||
Farbe<span class="addetailslist--detail--value" >
|
||||
{{ .Color }}</span>
|
||||
Art<span class="addetailslist--detail--value" >
|
||||
{{ .Type }}</span>
|
||||
</li>
|
||||
</ul>
|
||||
</div>
|
||||
@@ -251,11 +255,14 @@ type AdConfig struct {
|
||||
Price string
|
||||
Category string
|
||||
Condition string
|
||||
Type string
|
||||
Color string
|
||||
Created string
|
||||
Text string
|
||||
Images []string // files in ./t/
|
||||
}
|
||||
|
||||
// used to generate ad listings returned by httpmock using templates
|
||||
var adsrc = []AdConfig{
|
||||
{
|
||||
Title: "First Ad",
|
||||
@@ -263,7 +270,9 @@ var adsrc = []AdConfig{
|
||||
Category: "Klimbim",
|
||||
Text: "Thing to sale",
|
||||
Slug: "first-ad",
|
||||
Condition: "works",
|
||||
Condition: "Sehr Gut",
|
||||
Color: "Grün",
|
||||
Type: "Ball",
|
||||
Created: "Yesterday",
|
||||
Images: []string{"t/1.jpg", "t/2.jpg"},
|
||||
},
|
||||
@@ -273,7 +282,7 @@ var adsrc = []AdConfig{
|
||||
Category: "Kram",
|
||||
Text: "Thing to sale",
|
||||
Slug: "second-ad",
|
||||
Condition: "works",
|
||||
Condition: "Gut",
|
||||
Created: "Yesterday",
|
||||
Images: []string{"t/1.jpg", "t/2.jpg"},
|
||||
},
|
||||
@@ -284,7 +293,7 @@ var adsrc = []AdConfig{
|
||||
Category: "Kuddelmuddel",
|
||||
Text: "Thing to sale",
|
||||
Slug: "third-ad",
|
||||
Condition: "works",
|
||||
Condition: "In Ordnung",
|
||||
Created: "Yesterday",
|
||||
Images: []string{"t/1.jpg", "t/2.jpg"},
|
||||
},
|
||||
@@ -295,7 +304,7 @@ var adsrc = []AdConfig{
|
||||
Category: "Krempel",
|
||||
Text: "Thing to sale",
|
||||
Slug: "fourth-ad",
|
||||
Condition: "works",
|
||||
Condition: "Neu",
|
||||
Created: "Yesterday",
|
||||
Images: []string{"t/1.jpg", "t/2.jpg"},
|
||||
},
|
||||
@@ -306,7 +315,7 @@ var adsrc = []AdConfig{
|
||||
Category: "Kladderadatsch",
|
||||
Text: "Thing to sale",
|
||||
Slug: "fifth-ad",
|
||||
Condition: "works",
|
||||
Condition: "Sehr Gut",
|
||||
Created: "Yesterday",
|
||||
Images: []string{"t/1.jpg", "t/2.jpg"},
|
||||
},
|
||||
@@ -317,7 +326,7 @@ var adsrc = []AdConfig{
|
||||
Category: "Klunker",
|
||||
Text: "Thing to sale",
|
||||
Slug: "sixth-ad",
|
||||
Condition: "works",
|
||||
Condition: "Sehr Gut",
|
||||
Created: "Yesterday",
|
||||
Images: []string{"t/1.jpg", "t/2.jpg"},
|
||||
},
|
||||
@@ -328,7 +337,7 @@ var adsrc = []AdConfig{
|
||||
Category: "Klunker",
|
||||
Text: "Thing to sale",
|
||||
Slug: "seventh-ad",
|
||||
Condition: "works",
|
||||
Condition: "Sehr Gut",
|
||||
Created: "Yesterday",
|
||||
Images: []string{"t/1.png", "t/1.gif", "t/1.webp", "t/1.jpg"},
|
||||
},
|
||||
|
||||
12
scrape.go
12
scrape.go
@@ -22,6 +22,7 @@ import (
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"path/filepath"
|
||||
"slices"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -124,6 +125,17 @@ func ScrapeAd(fetch *Fetcher, uri string) error {
|
||||
return fmt.Errorf("could not extract ad data from page, got empty struct")
|
||||
}
|
||||
|
||||
for _, detail := range advertisement.Details {
|
||||
switch {
|
||||
case slices.Contains(CONDITIONS, detail):
|
||||
advertisement.Condition = detail
|
||||
case slices.Contains(COLORS, detail):
|
||||
advertisement.Color = detail
|
||||
default:
|
||||
advertisement.Type = detail
|
||||
}
|
||||
}
|
||||
|
||||
advertisement.CalculateExpire()
|
||||
|
||||
// prepare ad dir name
|
||||
|
||||
Reference in New Issue
Block a user