Compare commits
4 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e746b9cd31 | |||
| 50124af833 | |||
| 69a0149bbe | |||
| cf255e2380 |
@@ -5,6 +5,7 @@
|
||||
package handler
|
||||
|
||||
import (
|
||||
_ "embed"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io/fs"
|
||||
@@ -19,6 +20,9 @@ import (
|
||||
"github.com/baicai2026-baicai/goods/api/internal/store"
|
||||
)
|
||||
|
||||
//go:embed openapi.json
|
||||
var openAPISpec []byte
|
||||
|
||||
// APIVersion is the current public API version prefix.
|
||||
const APIVersion = "v1"
|
||||
|
||||
@@ -66,17 +70,22 @@ func (h *Handler) Router() http.Handler {
|
||||
r.Get("/healthz", h.Healthz)
|
||||
|
||||
r.Route("/api/"+APIVersion, func(r chi.Router) {
|
||||
r.Use(h.rateLimit)
|
||||
r.Route("/products", func(r chi.Router) {
|
||||
r.Get("/barcode/{gtin}", h.ProductByBarcode)
|
||||
r.Get("/search", h.SearchProducts)
|
||||
r.Get("/{id}", h.ProductByID)
|
||||
r.Get("/{id}/nutriments", h.ProductNutriments)
|
||||
r.Get("/{id}/msrp", h.ProductMSRP)
|
||||
// Machine-readable spec; not rate limited so tooling can always fetch it.
|
||||
r.Get("/openapi.json", h.OpenAPI)
|
||||
|
||||
r.Group(func(r chi.Router) {
|
||||
r.Use(h.rateLimit)
|
||||
r.Route("/products", func(r chi.Router) {
|
||||
r.Get("/barcode/{gtin}", h.ProductByBarcode)
|
||||
r.Get("/search", h.SearchProducts)
|
||||
r.Get("/{id}", h.ProductByID)
|
||||
r.Get("/{id}/nutriments", h.ProductNutriments)
|
||||
r.Get("/{id}/msrp", h.ProductMSRP)
|
||||
})
|
||||
r.Get("/brands", h.ListBrands)
|
||||
r.Get("/categories", h.ListCategories)
|
||||
r.Get("/sources/{id}", h.SourceByID)
|
||||
})
|
||||
r.Get("/brands", h.ListBrands)
|
||||
r.Get("/categories", h.ListCategories)
|
||||
r.Get("/sources/{id}", h.SourceByID)
|
||||
})
|
||||
|
||||
// Public SPA (homepage + search + contribute). API routes above take
|
||||
@@ -113,6 +122,12 @@ func (h *Handler) Healthz(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, map[string]string{"status": "ok"})
|
||||
}
|
||||
|
||||
// OpenAPI serves the embedded OpenAPI 3 specification for the public API.
|
||||
func (h *Handler) OpenAPI(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json; charset=utf-8")
|
||||
_, _ = w.Write(openAPISpec)
|
||||
}
|
||||
|
||||
// ProductByBarcode returns a product by its GTIN.
|
||||
func (h *Handler) ProductByBarcode(w http.ResponseWriter, r *http.Request) {
|
||||
p, err := h.store.ProductByGTIN(r.Context(), chi.URLParam(r, "gtin"))
|
||||
@@ -131,13 +146,19 @@ func (h *Handler) ProductByID(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, p)
|
||||
}
|
||||
|
||||
// SearchProducts runs a fuzzy name search with optional category filter + paging.
|
||||
// SearchProducts runs a trigram-fuzzy name search with optional
|
||||
// category/brand/country filters, ranked by relevance, plus paging.
|
||||
func (h *Handler) SearchProducts(w http.ResponseWriter, r *http.Request) {
|
||||
q := r.URL.Query().Get("q")
|
||||
category := r.URL.Query().Get("category")
|
||||
qv := r.URL.Query()
|
||||
filters := store.SearchFilters{
|
||||
Query: strings.TrimSpace(qv.Get("q")),
|
||||
Category: strings.TrimSpace(qv.Get("category")),
|
||||
Brand: strings.TrimSpace(qv.Get("brand")),
|
||||
Country: strings.TrimSpace(qv.Get("country")),
|
||||
}
|
||||
page, size := pageParams(r)
|
||||
|
||||
items, total, err := h.store.SearchProducts(r.Context(), q, category, size, (page-1)*size)
|
||||
items, total, err := h.store.SearchProducts(r.Context(), filters, size, (page-1)*size)
|
||||
if h.handleErr(w, r, err) {
|
||||
return
|
||||
}
|
||||
|
||||
@@ -116,6 +116,101 @@ func TestSearchProducts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestSearchFuzzyAndFilters(t *testing.T) {
|
||||
h, _ := newTestHandler(t)
|
||||
ctx := context.Background()
|
||||
dsn := os.Getenv("OPENGOODS_DATABASE_URL")
|
||||
if dsn == "" {
|
||||
dsn = "postgres://opengoods:opengoods@localhost:5432/opengoods?sslmode=disable"
|
||||
}
|
||||
pool, err := pgxpool.New(ctx, dsn)
|
||||
if err != nil {
|
||||
t.Skipf("no database: %v", err)
|
||||
}
|
||||
defer pool.Close()
|
||||
|
||||
_, err = pool.Exec(ctx,
|
||||
"INSERT INTO brand (name, normalized_name) VALUES ('ZZ Test Brand','zz test brand') ON CONFLICT DO NOTHING")
|
||||
if err != nil {
|
||||
t.Fatalf("seed brand: %v", err)
|
||||
}
|
||||
_, err = pool.Exec(ctx, `
|
||||
INSERT INTO product (name, brand_id, country_of_origin, quality_score, status)
|
||||
VALUES ('ZZ Hazelnut Chocolate', (SELECT id FROM brand WHERE name='ZZ Test Brand'), 'Testland', 0.5, 'active')`)
|
||||
if err != nil {
|
||||
t.Fatalf("seed product: %v", err)
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
_, _ = pool.Exec(ctx, "DELETE FROM product WHERE name='ZZ Hazelnut Chocolate'")
|
||||
_, _ = pool.Exec(ctx, "DELETE FROM brand WHERE name='ZZ Test Brand'")
|
||||
})
|
||||
|
||||
decode := func(path string) []store.ProductSummary {
|
||||
rec := doGET(t, h, path)
|
||||
if rec.Code != http.StatusOK {
|
||||
t.Fatalf("%s -> status %d", path, rec.Code)
|
||||
}
|
||||
var body struct {
|
||||
Items []store.ProductSummary `json:"items"`
|
||||
}
|
||||
if err := json.NewDecoder(rec.Body).Decode(&body); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return body.Items
|
||||
}
|
||||
has := func(items []store.ProductSummary, name string) *store.ProductSummary {
|
||||
for i := range items {
|
||||
if items[i].Name == name {
|
||||
return &items[i]
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Typo "choclate" should fuzzy-match via word_similarity and carry a score.
|
||||
got := has(decode("/api/"+APIVersion+"/products/search?q=choclate"), "ZZ Hazelnut Chocolate")
|
||||
if got == nil {
|
||||
t.Fatal("fuzzy query 'choclate' did not match 'ZZ Hazelnut Chocolate'")
|
||||
}
|
||||
if got.Score == nil || *got.Score <= 0 {
|
||||
t.Fatalf("expected positive fuzzy score, got %v", got.Score)
|
||||
}
|
||||
|
||||
// Brand filter.
|
||||
if has(decode("/api/"+APIVersion+"/products/search?brand=ZZ+Test+Brand"), "ZZ Hazelnut Chocolate") == nil {
|
||||
t.Fatal("brand filter did not return the product")
|
||||
}
|
||||
// Country filter (case-insensitive prefix).
|
||||
if has(decode("/api/"+APIVersion+"/products/search?country=test"), "ZZ Hazelnut Chocolate") == nil {
|
||||
t.Fatal("country filter did not return the product")
|
||||
}
|
||||
// Non-matching country excludes it.
|
||||
if has(decode("/api/"+APIVersion+"/products/search?country=france"), "ZZ Hazelnut Chocolate") != nil {
|
||||
t.Fatal("country filter 'france' should not return the product")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOpenAPISpec(t *testing.T) {
|
||||
h, _ := newTestHandler(t)
|
||||
rec := doGET(t, h, "/api/"+APIVersion+"/openapi.json")
|
||||
if rec.Code != http.StatusOK {
|
||||
t.Fatalf("status = %d", rec.Code)
|
||||
}
|
||||
var spec struct {
|
||||
OpenAPI string `json:"openapi"`
|
||||
Paths map[string]any `json:"paths"`
|
||||
}
|
||||
if err := json.NewDecoder(rec.Body).Decode(&spec); err != nil {
|
||||
t.Fatalf("openapi.json is not valid JSON: %v", err)
|
||||
}
|
||||
if spec.OpenAPI == "" || len(spec.Paths) == 0 {
|
||||
t.Fatalf("unexpected spec: %+v", spec)
|
||||
}
|
||||
if _, ok := spec.Paths["/products/search"]; !ok {
|
||||
t.Fatal("spec missing /products/search path")
|
||||
}
|
||||
}
|
||||
|
||||
func TestListCategories(t *testing.T) {
|
||||
h, _ := newTestHandler(t)
|
||||
rec := doGET(t, h, "/api/"+APIVersion+"/categories")
|
||||
|
||||
@@ -0,0 +1,186 @@
|
||||
{
|
||||
"openapi": "3.0.3",
|
||||
"info": {
|
||||
"title": "OpenGoods / 天工商品档案公共仓 API",
|
||||
"version": "1.0.0",
|
||||
"description": "Public, read-only product-facts REST API. Anonymous access is allowed at a lower per-minute rate; an optional API key grants a higher rate limit and attributes usage. No purchase or commerce endpoints by design.",
|
||||
"license": { "name": "Data under each source's license (e.g. ODbL)" }
|
||||
},
|
||||
"servers": [{ "url": "https://goods.tangshasha.com/api/v1" }],
|
||||
"tags": [
|
||||
{ "name": "products" },
|
||||
{ "name": "catalog" },
|
||||
{ "name": "meta" }
|
||||
],
|
||||
"security": [{ "ApiKeyHeader": [] }, { "BearerKey": [] }, {}],
|
||||
"paths": {
|
||||
"/products/search": {
|
||||
"get": {
|
||||
"tags": ["products"],
|
||||
"summary": "Search products",
|
||||
"description": "Trigram-fuzzy name search (typo-tolerant) with optional category/brand/country filters, ranked by name similarity blended with data quality_score.",
|
||||
"parameters": [
|
||||
{ "name": "q", "in": "query", "schema": { "type": "string" }, "description": "Keyword (name or barcode); fuzzy-matched. Empty returns all, ordered by quality_score." },
|
||||
{ "name": "category", "in": "query", "schema": { "type": "string" }, "description": "Category code (matches the subtree), e.g. food.beverages." },
|
||||
{ "name": "brand", "in": "query", "schema": { "type": "string" }, "description": "Brand name (fuzzy)." },
|
||||
{ "name": "country", "in": "query", "schema": { "type": "string" }, "description": "Country of origin (case-insensitive prefix)." },
|
||||
{ "name": "page", "in": "query", "schema": { "type": "integer", "default": 1, "minimum": 1 } },
|
||||
{ "name": "size", "in": "query", "schema": { "type": "integer", "default": 20, "maximum": 100 } }
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Paged search results.",
|
||||
"headers": {
|
||||
"X-RateLimit-Limit": { "schema": { "type": "integer" }, "description": "Max requests in the current window." },
|
||||
"X-RateLimit-Remaining": { "schema": { "type": "integer" }, "description": "Remaining requests in the window." },
|
||||
"X-RateLimit-Reset": { "schema": { "type": "integer" }, "description": "Unix timestamp when the window resets." }
|
||||
},
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"items": { "type": "array", "items": { "$ref": "#/components/schemas/ProductSummary" } },
|
||||
"page": { "type": "integer" },
|
||||
"size": { "type": "integer" },
|
||||
"total": { "type": "integer" }
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"401": { "$ref": "#/components/responses/InvalidApiKey" },
|
||||
"429": { "$ref": "#/components/responses/RateLimited" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"/products/barcode/{gtin}": {
|
||||
"get": {
|
||||
"tags": ["products"],
|
||||
"summary": "Get product by barcode (GTIN)",
|
||||
"parameters": [{ "name": "gtin", "in": "path", "required": true, "schema": { "type": "string" } }],
|
||||
"responses": {
|
||||
"200": { "description": "Product", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Product" } } } },
|
||||
"404": { "$ref": "#/components/responses/NotFound" },
|
||||
"429": { "$ref": "#/components/responses/RateLimited" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"/products/{id}": {
|
||||
"get": {
|
||||
"tags": ["products"],
|
||||
"summary": "Get product detail by UUID",
|
||||
"parameters": [{ "name": "id", "in": "path", "required": true, "schema": { "type": "string", "format": "uuid" } }],
|
||||
"responses": {
|
||||
"200": { "description": "Product", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/Product" } } } },
|
||||
"404": { "$ref": "#/components/responses/NotFound" }
|
||||
}
|
||||
}
|
||||
},
|
||||
"/products/{id}/nutriments": {
|
||||
"get": {
|
||||
"tags": ["products"],
|
||||
"summary": "Get product nutriments",
|
||||
"parameters": [{ "name": "id", "in": "path", "required": true, "schema": { "type": "string", "format": "uuid" } }],
|
||||
"responses": { "200": { "description": "Nutriments" }, "404": { "$ref": "#/components/responses/NotFound" } }
|
||||
}
|
||||
},
|
||||
"/products/{id}/msrp": {
|
||||
"get": {
|
||||
"tags": ["products"],
|
||||
"summary": "Get manufacturer suggested retail price snapshots (reference only)",
|
||||
"parameters": [{ "name": "id", "in": "path", "required": true, "schema": { "type": "string", "format": "uuid" } }],
|
||||
"responses": { "200": { "description": "MSRP snapshots" } }
|
||||
}
|
||||
},
|
||||
"/brands": {
|
||||
"get": {
|
||||
"tags": ["catalog"],
|
||||
"summary": "List brands",
|
||||
"parameters": [
|
||||
{ "name": "page", "in": "query", "schema": { "type": "integer", "default": 1 } },
|
||||
{ "name": "size", "in": "query", "schema": { "type": "integer", "default": 20, "maximum": 100 } }
|
||||
],
|
||||
"responses": { "200": { "description": "Paged brands" } }
|
||||
}
|
||||
},
|
||||
"/categories": {
|
||||
"get": { "tags": ["catalog"], "summary": "List the category tree", "responses": { "200": { "description": "Category tree" } } }
|
||||
},
|
||||
"/sources/{id}": {
|
||||
"get": {
|
||||
"tags": ["meta"],
|
||||
"summary": "Get a data source",
|
||||
"parameters": [{ "name": "id", "in": "path", "required": true, "schema": { "type": "string", "format": "uuid" } }],
|
||||
"responses": { "200": { "description": "Source" }, "404": { "$ref": "#/components/responses/NotFound" } }
|
||||
}
|
||||
}
|
||||
},
|
||||
"components": {
|
||||
"securitySchemes": {
|
||||
"ApiKeyHeader": { "type": "apiKey", "in": "header", "name": "X-API-Key", "description": "API key, e.g. og_live_xxx. Optional." },
|
||||
"BearerKey": { "type": "http", "scheme": "bearer", "description": "Authorization: Bearer og_live_xxx. Optional." }
|
||||
},
|
||||
"responses": {
|
||||
"NotFound": {
|
||||
"description": "Resource not found.",
|
||||
"content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
|
||||
},
|
||||
"RateLimited": {
|
||||
"description": "Rate limit exceeded.",
|
||||
"headers": { "Retry-After": { "schema": { "type": "integer" }, "description": "Seconds to wait before retrying." } },
|
||||
"content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
|
||||
},
|
||||
"InvalidApiKey": {
|
||||
"description": "API key invalid or revoked.",
|
||||
"content": { "application/json": { "schema": { "$ref": "#/components/schemas/Error" } } }
|
||||
}
|
||||
},
|
||||
"schemas": {
|
||||
"Error": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"error": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"code": { "type": "string" },
|
||||
"message": { "type": "string" },
|
||||
"request_id": { "type": "string" }
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"ProductSummary": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"id": { "type": "string", "format": "uuid" },
|
||||
"gtin": { "type": "string", "nullable": true },
|
||||
"name": { "type": "string" },
|
||||
"brand": { "type": "string", "nullable": true },
|
||||
"category_path": { "type": "string", "nullable": true },
|
||||
"country_of_origin": { "type": "string", "nullable": true },
|
||||
"quality_score": { "type": "number", "format": "float" },
|
||||
"score": { "type": "number", "format": "float", "nullable": true, "description": "Relevance (name word-similarity) when q is provided; null otherwise." }
|
||||
}
|
||||
},
|
||||
"Product": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"id": { "type": "string", "format": "uuid" },
|
||||
"gtin": { "type": "string", "nullable": true },
|
||||
"name": { "type": "string" },
|
||||
"brand": { "type": "string", "nullable": true },
|
||||
"category_path": { "type": "string", "nullable": true },
|
||||
"net_content_value": { "type": "number", "nullable": true },
|
||||
"net_content_unit": { "type": "string", "nullable": true },
|
||||
"country_of_origin": { "type": "string", "nullable": true },
|
||||
"quality_score": { "type": "number", "format": "float" },
|
||||
"nutriments": { "type": "object", "additionalProperties": true, "nullable": true },
|
||||
"nutrition_basis": { "type": "string", "nullable": true },
|
||||
"nutri_score": { "type": "string", "nullable": true },
|
||||
"ingredients_text": { "type": "string", "nullable": true }
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+67
-22
@@ -82,11 +82,22 @@ func (s *Store) ProductBarcodes(ctx context.Context, productID string) ([]Barcod
|
||||
|
||||
// ProductSummary is a lightweight row used in search/listing responses.
|
||||
type ProductSummary struct {
|
||||
ID string `json:"id"`
|
||||
GTIN *string `json:"gtin"`
|
||||
Name string `json:"name"`
|
||||
Brand *string `json:"brand"`
|
||||
CategoryPath *string `json:"category_path"`
|
||||
ID string `json:"id"`
|
||||
GTIN *string `json:"gtin"`
|
||||
Name string `json:"name"`
|
||||
Brand *string `json:"brand"`
|
||||
CategoryPath *string `json:"category_path"`
|
||||
Country *string `json:"country_of_origin"`
|
||||
QualityScore float64 `json:"quality_score"`
|
||||
Score *float64 `json:"score,omitempty"`
|
||||
}
|
||||
|
||||
// SearchFilters bundles the optional filters accepted by SearchProducts.
|
||||
type SearchFilters struct {
|
||||
Query string // fuzzy name / barcode query
|
||||
Category string // ltree path; matches the subtree
|
||||
Brand string // fuzzy brand name
|
||||
Country string // country_of_origin prefix (case-insensitive)
|
||||
}
|
||||
|
||||
const productSelect = `
|
||||
@@ -148,34 +159,67 @@ func (s *Store) ProductByID(ctx context.Context, id string) (*Product, error) {
|
||||
return p, nil
|
||||
}
|
||||
|
||||
// SearchProducts performs a fuzzy name search with optional category subtree filter.
|
||||
func (s *Store) SearchProducts(ctx context.Context, q, category string, limit, offset int) ([]ProductSummary, int, error) {
|
||||
// fuzzyThreshold is the minimum word_similarity for a name to be considered a
|
||||
// fuzzy match. ~0.42 tolerates common typos (e.g. "choclate"→"Chocolate")
|
||||
// without returning unrelated products.
|
||||
const fuzzyThreshold = "0.42"
|
||||
|
||||
// SearchProducts runs a trigram-fuzzy name search with optional category /
|
||||
// brand / country filters. When a query is present, matching is inclusive
|
||||
// (substring OR trigram-similar OR barcode), and results are ranked by name
|
||||
// similarity blended with quality_score so the best, most-complete records
|
||||
// surface first. Without a query, results are ordered by quality_score.
|
||||
func (s *Store) SearchProducts(ctx context.Context, f SearchFilters, limit, offset int) ([]ProductSummary, int, error) {
|
||||
args := []any{}
|
||||
where := "WHERE p.status = 'active'"
|
||||
if q != "" {
|
||||
args = append(args, q)
|
||||
where += ` AND (p.name ILIKE '%' || $1 || '%'
|
||||
|
||||
qIdx := 0
|
||||
if f.Query != "" {
|
||||
args = append(args, f.Query)
|
||||
qIdx = len(args)
|
||||
q := "$" + strconv.Itoa(qIdx)
|
||||
where += ` AND (p.name ILIKE '%' || ` + q + ` || '%'
|
||||
OR word_similarity(` + q + `, p.name) >= ` + fuzzyThreshold + `
|
||||
OR EXISTS (SELECT 1 FROM product_barcode pb
|
||||
WHERE pb.product_id = p.id AND pb.gtin ILIKE '%' || $1 || '%'))`
|
||||
WHERE pb.product_id = p.id AND pb.gtin ILIKE '%' || ` + q + ` || '%'))`
|
||||
}
|
||||
if category != "" {
|
||||
args = append(args, category)
|
||||
if f.Category != "" {
|
||||
args = append(args, f.Category)
|
||||
where += " AND c.path <@ $" + strconv.Itoa(len(args)) + "::ltree"
|
||||
}
|
||||
if f.Brand != "" {
|
||||
args = append(args, f.Brand)
|
||||
where += " AND b.name ILIKE '%' || $" + strconv.Itoa(len(args)) + " || '%'"
|
||||
}
|
||||
if f.Country != "" {
|
||||
args = append(args, f.Country)
|
||||
where += " AND p.country_of_origin ILIKE $" + strconv.Itoa(len(args)) + " || '%'"
|
||||
}
|
||||
|
||||
from := `FROM product p
|
||||
LEFT JOIN brand b ON b.id = p.brand_id
|
||||
LEFT JOIN category c ON c.id = p.category_id `
|
||||
|
||||
countSQL := "SELECT count(*) FROM product p LEFT JOIN category c ON c.id = p.category_id " + where
|
||||
var total int
|
||||
if err := s.pool.QueryRow(ctx, countSQL, args...).Scan(&total); err != nil {
|
||||
if err := s.pool.QueryRow(ctx, "SELECT count(*) "+from+where, args...).Scan(&total); err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
|
||||
// Ranking: when querying, similarity drives order, multiplied by a
|
||||
// quality factor floored at 0.5 so low-quality records aren't zeroed out.
|
||||
scoreExpr := "NULL::real"
|
||||
orderBy := "p.quality_score DESC, p.name"
|
||||
if f.Query != "" {
|
||||
q := "$" + strconv.Itoa(qIdx)
|
||||
scoreExpr = "word_similarity(" + q + ", p.name)"
|
||||
orderBy = scoreExpr + " * (0.5 + p.quality_score) DESC, p.quality_score DESC, p.name"
|
||||
}
|
||||
|
||||
args = append(args, limit, offset)
|
||||
listSQL := `
|
||||
SELECT p.id, p.gtin, p.name, b.name, c.path::text
|
||||
FROM product p
|
||||
LEFT JOIN brand b ON b.id = p.brand_id
|
||||
LEFT JOIN category c ON c.id = p.category_id ` + where +
|
||||
" ORDER BY p.name LIMIT $" + strconv.Itoa(len(args)-1) + " OFFSET $" + strconv.Itoa(len(args))
|
||||
listSQL := "SELECT p.id, p.gtin, p.name, b.name, c.path::text, p.country_of_origin, p.quality_score, " +
|
||||
scoreExpr + " AS score " + from + where +
|
||||
" ORDER BY " + orderBy +
|
||||
" LIMIT $" + strconv.Itoa(len(args)-1) + " OFFSET $" + strconv.Itoa(len(args))
|
||||
|
||||
rows, err := s.pool.Query(ctx, listSQL, args...)
|
||||
if err != nil {
|
||||
@@ -186,7 +230,8 @@ LEFT JOIN category c ON c.id = p.category_id ` + where +
|
||||
out := []ProductSummary{}
|
||||
for rows.Next() {
|
||||
var ps ProductSummary
|
||||
if err := rows.Scan(&ps.ID, &ps.GTIN, &ps.Name, &ps.Brand, &ps.CategoryPath); err != nil {
|
||||
if err := rows.Scan(&ps.ID, &ps.GTIN, &ps.Name, &ps.Brand, &ps.CategoryPath,
|
||||
&ps.Country, &ps.QualityScore, &ps.Score); err != nil {
|
||||
return nil, 0, err
|
||||
}
|
||||
out = append(out, ps)
|
||||
|
||||
+117
@@ -0,0 +1,117 @@
|
||||
# OpenGoods 公共 API 开发者文档
|
||||
|
||||
天工商品档案公共仓(OpenGoods)提供**公开、只读**的商品事实 REST API:按条码/名称查询商品的客观资料(品牌、品类、净含量、产地、配料、营养成分、Nutri-Score、厂商建议零售价快照等)。返回均为 JSON(UTF-8)。**本服务不含任何购买/交易接口。**
|
||||
|
||||
- 基础地址:`https://goods.tangshasha.com/api/v1`
|
||||
- 机器可读规范(OpenAPI 3):`GET /api/v1/openapi.json`
|
||||
- 交互式文档:站点「API 调用说明」页
|
||||
|
||||
## 鉴权
|
||||
|
||||
API 默认**匿名可用**,无需任何凭证即可调用。匿名请求按来源 IP 计入一个较低的默认每分钟额度。
|
||||
|
||||
如需更高额度并让用量归属到你,可向运营方申请一枚 **API Key**(形如 `og_live_xxxxxxxx`),请求时二选一携带:
|
||||
|
||||
```bash
|
||||
curl -H "X-API-Key: og_live_xxxxxxxx" \
|
||||
"https://goods.tangshasha.com/api/v1/products/search?q=牛奶"
|
||||
|
||||
# 或
|
||||
curl -H "Authorization: Bearer og_live_xxxxxxxx" \
|
||||
"https://goods.tangshasha.com/api/v1/products/search?q=牛奶"
|
||||
```
|
||||
|
||||
> 仅在创建时返回一次明文 Key,请妥善保存。服务端只存储其 SHA-256 哈希。
|
||||
|
||||
## 限流
|
||||
|
||||
采用**固定窗口**限流(每分钟)。每个响应都会回写以下响应头:
|
||||
|
||||
| 响应头 | 含义 |
|
||||
| --- | --- |
|
||||
| `X-RateLimit-Limit` | 当前窗口允许的最大请求数 |
|
||||
| `X-RateLimit-Remaining` | 当前窗口剩余可用次数 |
|
||||
| `X-RateLimit-Reset` | 窗口重置的 Unix 时间戳(秒) |
|
||||
| `Retry-After` | 仅在超额(429)时返回,建议等待的秒数 |
|
||||
|
||||
- 超过额度:`429 Too Many Requests`,错误码 `rate_limited`。
|
||||
- Key 无效或已吊销:`401 Unauthorized`,错误码 `invalid_api_key`。
|
||||
|
||||
## 错误格式
|
||||
|
||||
非 2xx 响应体统一为:
|
||||
|
||||
```json
|
||||
{ "error": { "code": "not_found", "message": "…", "request_id": "…" } }
|
||||
```
|
||||
|
||||
## 分页
|
||||
|
||||
列表类接口支持 `page`(默认 `1`)与 `size`(默认 `20`,最大 `100`),响应含 `page`/`size`/`total`。
|
||||
|
||||
## 端点
|
||||
|
||||
### `GET /products/search` — 搜索商品
|
||||
|
||||
按名称做三元组(trigram)模糊搜索,**可容忍错别字**;支持品类/品牌/产地过滤;结果按相关度(名称相似度 × 数据质量分)排序。
|
||||
|
||||
| 参数 | 必填 | 说明 |
|
||||
| --- | --- | --- |
|
||||
| `q` | 否 | 关键词(名称/条码),模糊匹配;留空则按质量分返回全部 |
|
||||
| `category` | 否 | 品类编码(含子树),如 `food.beverages` |
|
||||
| `brand` | 否 | 品牌名(模糊匹配),如 `Ferrero` |
|
||||
| `country` | 否 | 产地前缀(不区分大小写),如 `China` |
|
||||
| `page` | 否 | 页码,默认 1 |
|
||||
| `size` | 否 | 每页条数,默认 20,最大 100 |
|
||||
|
||||
```bash
|
||||
curl "https://goods.tangshasha.com/api/v1/products/search?q=nutela&country=Italy"
|
||||
```
|
||||
|
||||
```json
|
||||
{
|
||||
"items": [
|
||||
{
|
||||
"id": "…",
|
||||
"gtin": "3017624010701",
|
||||
"name": "Nutella",
|
||||
"brand": "Ferrero",
|
||||
"category_path": "food.snacks.chocolate",
|
||||
"country_of_origin": "Italy",
|
||||
"quality_score": 0.81,
|
||||
"score": 0.71
|
||||
}
|
||||
],
|
||||
"page": 1,
|
||||
"size": 20,
|
||||
"total": 1
|
||||
}
|
||||
```
|
||||
|
||||
`score` 为名称相关度(提供 `q` 时返回,0–1),未提供 `q` 时为 `null`。
|
||||
|
||||
### `GET /products/barcode/{gtin}` — 按条码查询
|
||||
|
||||
```bash
|
||||
curl "https://goods.tangshasha.com/api/v1/products/barcode/5449000000996"
|
||||
```
|
||||
|
||||
### `GET /products/{id}` — 商品详情
|
||||
|
||||
按商品 UUID 获取完整档案(含配料、营养、添加剂、图片、MSRP 等)。
|
||||
|
||||
### `GET /products/{id}/nutriments` — 商品营养成分
|
||||
|
||||
### `GET /products/{id}/msrp` — 厂商建议零售价快照
|
||||
|
||||
官方建议零售价历史快照,仅供参考,不含任何购买入口。
|
||||
|
||||
### `GET /brands` — 品牌列表(分页)
|
||||
|
||||
### `GET /categories` — 品类树
|
||||
|
||||
### `GET /sources/{id}` — 数据来源
|
||||
|
||||
## 免责声明
|
||||
|
||||
数据可能存在误差或滞后,按「现状」提供,不构成医疗/购买建议。商品资料版权归各原始来源所有,请遵循其许可(如 OpenFoodFacts 的 ODbL),引用时请注明天工商品档案公共仓及原始来源。
|
||||
@@ -0,0 +1,2 @@
|
||||
DROP INDEX IF EXISTS idx_product_country;
|
||||
DROP INDEX IF EXISTS idx_brand_name_trgm;
|
||||
@@ -0,0 +1,9 @@
|
||||
-- Search upgrade: trigram-based fuzzy matching + brand/country filters.
|
||||
-- pg_trgm and the product.name GIN index already exist (see 0001). Add a
|
||||
-- matching trigram index on brand.name so brand filtering / fuzzy brand
|
||||
-- lookups can use an index instead of a sequential scan.
|
||||
CREATE INDEX IF NOT EXISTS idx_brand_name_trgm ON brand USING gin (name gin_trgm_ops);
|
||||
|
||||
-- country_of_origin is filtered by exact/prefix match; a plain btree index
|
||||
-- keeps that cheap as the catalog grows.
|
||||
CREATE INDEX IF NOT EXISTS idx_product_country ON product (country_of_origin);
|
||||
@@ -25,11 +25,22 @@ export interface SearchResult {
|
||||
total: number;
|
||||
}
|
||||
|
||||
export interface SearchFilters {
|
||||
brand?: string;
|
||||
country?: string;
|
||||
}
|
||||
|
||||
export const api = {
|
||||
search: (q: string, page = 1, size = 20) =>
|
||||
req<SearchResult>(
|
||||
`/api/v1/products/search?q=${encodeURIComponent(q)}&page=${page}&size=${size}`,
|
||||
),
|
||||
search: (q: string, page = 1, size = 20, filters: SearchFilters = {}) => {
|
||||
const params = new URLSearchParams({
|
||||
q,
|
||||
page: String(page),
|
||||
size: String(size),
|
||||
});
|
||||
if (filters.brand) params.set("brand", filters.brand);
|
||||
if (filters.country) params.set("country", filters.country);
|
||||
return req<SearchResult>(`/api/v1/products/search?${params.toString()}`);
|
||||
},
|
||||
product: (id: string) => req<Product>(`/api/v1/products/${id}`),
|
||||
categories: () => req<{ items: Category[] }>(`/api/v1/categories`),
|
||||
submit: (input: SubmissionInput) =>
|
||||
|
||||
@@ -117,7 +117,7 @@ export default function ApiDocs() {
|
||||
基础地址:<code className="font-mono bg-gray-100 rounded px-1.5 py-0.5">{BASE}</code>
|
||||
</div>
|
||||
<ul className="mt-2 list-disc pl-5 text-gray-600 space-y-1">
|
||||
<li>无需 API Key / Token,直接 GET 即可。</li>
|
||||
<li>无需 API Key / Token 即可直接 GET;带 Key 可获得更高频率上限(见下文「鉴权与限流」)。</li>
|
||||
<li>
|
||||
分页参数 <code className="font-mono">page</code>(默认 1)、
|
||||
<code className="font-mono">size</code>(默认 20,最大 100)。
|
||||
@@ -131,6 +131,53 @@ export default function ApiDocs() {
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="bg-white border rounded-lg p-5">
|
||||
<h2 className="text-lg font-semibold text-gray-800">鉴权与限流</h2>
|
||||
<p className="mt-2 text-gray-600 text-sm leading-relaxed">
|
||||
API 默认<strong>匿名可用</strong>:无需任何凭证即可调用,按来源 IP 计入一个较低的默认频率额度。
|
||||
如需更高额度并让用量归属到你,可在运营方申请一枚 API Key,请求时通过请求头携带:
|
||||
</p>
|
||||
<div className="mt-3">
|
||||
<Code>{`# 二选一
|
||||
curl -H "X-API-Key: og_live_xxxxxxxx" ${BASE}/products/search?q=牛奶
|
||||
curl -H "Authorization: Bearer og_live_xxxxxxxx" ${BASE}/products/search?q=牛奶`}</Code>
|
||||
</div>
|
||||
<p className="mt-3 text-gray-600 text-sm leading-relaxed">
|
||||
采用<strong>固定窗口</strong>限流(每分钟)。每个响应都会回写以下响应头,便于客户端自适应:
|
||||
</p>
|
||||
<table className="mt-3 w-full text-sm">
|
||||
<thead className="text-gray-400 text-left">
|
||||
<tr>
|
||||
<th className="font-medium pr-4 pb-1">响应头</th>
|
||||
<th className="font-medium pb-1">含义</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody className="align-top">
|
||||
<tr>
|
||||
<td className="pr-4 py-0.5 font-mono text-gray-700">X-RateLimit-Limit</td>
|
||||
<td className="py-0.5 text-gray-600">当前窗口允许的最大请求数</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td className="pr-4 py-0.5 font-mono text-gray-700">X-RateLimit-Remaining</td>
|
||||
<td className="py-0.5 text-gray-600">当前窗口剩余可用次数</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td className="pr-4 py-0.5 font-mono text-gray-700">X-RateLimit-Reset</td>
|
||||
<td className="py-0.5 text-gray-600">窗口重置的 Unix 时间戳(秒)</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td className="pr-4 py-0.5 font-mono text-gray-700">Retry-After</td>
|
||||
<td className="py-0.5 text-gray-600">超额时返回,建议等待的秒数</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
<p className="mt-3 text-gray-600 text-sm leading-relaxed">
|
||||
超过额度返回 <code className="font-mono">429 Too Many Requests</code>,错误码
|
||||
<code className="font-mono">rate_limited</code>;无效或已吊销的 Key 返回
|
||||
<code className="font-mono">401</code>,错误码 <code className="font-mono">invalid_api_key</code>。
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<Endpoint
|
||||
method="GET"
|
||||
path="/healthz"
|
||||
@@ -164,19 +211,24 @@ export default function ApiDocs() {
|
||||
method="GET"
|
||||
path="/api/v1/products/search"
|
||||
title="搜索商品"
|
||||
desc="按名称模糊搜索,可按品类过滤,支持分页。"
|
||||
desc="按名称做三元组(trigram)模糊搜索,可容忍错别字;支持品类/品牌/产地过滤;结果按相关度(名称相似度 × 数据质量分)排序。"
|
||||
params={[
|
||||
{ name: "q", desc: "关键词(名称/条码),留空返回全部" },
|
||||
{ name: "category", desc: "品类编码过滤,如 food.beverages" },
|
||||
{ name: "q", desc: "关键词(名称/条码),支持模糊匹配;留空则按质量分返回全部" },
|
||||
{ name: "category", desc: "品类编码(含子树),如 food.beverages" },
|
||||
{ name: "brand", desc: "品牌名(模糊匹配),如 Ferrero" },
|
||||
{ name: "country", desc: "产地前缀(不区分大小写),如 China" },
|
||||
{ name: "page", desc: "页码,默认 1" },
|
||||
{ name: "size", desc: "每页条数,默认 20,最大 100" },
|
||||
]}
|
||||
example={`curl "${BASE}/products/search?q=nutella&page=1&size=20"`}
|
||||
example={`curl "${BASE}/products/search?q=nutela&country=Italy&page=1&size=20"`}
|
||||
response={`{
|
||||
"items": [
|
||||
{ "id": "…", "gtin": "3017624010701",
|
||||
"name": "Nutella", "brand": "Ferrero",
|
||||
"category_path": "food.snacks.chocolate" }
|
||||
"category_path": "food.snacks.chocolate",
|
||||
"country_of_origin": "Italy",
|
||||
"quality_score": 0.81,
|
||||
"score": 0.71 }
|
||||
],
|
||||
"page": 1, "size": 20, "total": 1
|
||||
}`}
|
||||
|
||||
@@ -13,6 +13,8 @@ export default function Home({
|
||||
onApi: () => void;
|
||||
}) {
|
||||
const [q, setQ] = useState("");
|
||||
const [brand, setBrand] = useState("");
|
||||
const [country, setCountry] = useState("");
|
||||
const [items, setItems] = useState<ProductSummary[]>([]);
|
||||
const [total, setTotal] = useState(0);
|
||||
const [searched, setSearched] = useState(false);
|
||||
@@ -24,7 +26,10 @@ export default function Home({
|
||||
setLoading(true);
|
||||
setError("");
|
||||
try {
|
||||
const res = await api.search(q.trim(), 1, 30);
|
||||
const res = await api.search(q.trim(), 1, 30, {
|
||||
brand: brand.trim() || undefined,
|
||||
country: country.trim() || undefined,
|
||||
});
|
||||
setItems(res.items);
|
||||
setTotal(res.total);
|
||||
setSearched(true);
|
||||
@@ -61,6 +66,20 @@ export default function Home({
|
||||
{loading ? "检索中…" : "检索"}
|
||||
</button>
|
||||
</form>
|
||||
<div className="mt-3 max-w-2xl mx-auto flex flex-wrap items-center justify-center gap-2 text-sm">
|
||||
<input
|
||||
value={brand}
|
||||
onChange={(e) => setBrand(e.target.value)}
|
||||
placeholder="按品牌筛选(如 Ferrero)"
|
||||
className="flex-1 min-w-[14rem] bg-white border rounded-lg px-3 py-2 outline-none focus:ring-2 focus:ring-emerald-300"
|
||||
/>
|
||||
<input
|
||||
value={country}
|
||||
onChange={(e) => setCountry(e.target.value)}
|
||||
placeholder="按产地筛选(如 China)"
|
||||
className="flex-1 min-w-[14rem] bg-white border rounded-lg px-3 py-2 outline-none focus:ring-2 focus:ring-emerald-300"
|
||||
/>
|
||||
</div>
|
||||
<button
|
||||
onClick={onApi}
|
||||
className="mt-4 inline-flex items-center gap-1.5 text-sm text-emerald-700 hover:underline"
|
||||
@@ -105,7 +124,12 @@ export default function Home({
|
||||
{p.gtin ? ` · ${p.gtin}` : ""}
|
||||
</div>
|
||||
</div>
|
||||
<span className="text-xs text-gray-400">{p.category_path || ""}</span>
|
||||
<span className="text-xs text-gray-400 text-right">
|
||||
<span className="block">{p.category_path || ""}</span>
|
||||
{p.country_of_origin ? (
|
||||
<span className="block text-gray-400">产地:{p.country_of_origin}</span>
|
||||
) : null}
|
||||
</span>
|
||||
</button>
|
||||
</li>
|
||||
))}
|
||||
|
||||
@@ -4,6 +4,9 @@ export interface ProductSummary {
|
||||
name: string;
|
||||
brand: string | null;
|
||||
category_path: string | null;
|
||||
country_of_origin?: string | null;
|
||||
quality_score?: number;
|
||||
score?: number | null;
|
||||
}
|
||||
|
||||
export interface Barcode {
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
# build artifacts
|
||||
bypos-collector.exe
|
||||
bypos-collector
|
||||
bypos-collector-linux
|
||||
*.exe
|
||||
# collected data / outputs
|
||||
*.jsonl
|
||||
*.csv
|
||||
*.log
|
||||
@@ -0,0 +1,67 @@
|
||||
# bypos-collector
|
||||
|
||||
按条码批量采集商品档案的小工具(单文件 Windows/Linux 程序,自带本地 Web 控制台)。
|
||||
数据来源是云店「新增商品」输入条码时所查的同一个中心商品库 `zc.bypos.net`。
|
||||
采集结果落地为 JSONL,供后续导入本项目(天工/goods)。
|
||||
|
||||
> 这是一个**独立模块**(有自己的 `go.mod`),与 `api/` 主服务互不影响,
|
||||
> CI 不会编译它。放在 `tools/` 下仅作代码留存与后期迭代。
|
||||
|
||||
## 目录
|
||||
|
||||
| 文件 | 说明 |
|
||||
| --- | --- |
|
||||
| `main.go` | 入口:本地 HTTP 服务 + 启动浏览器 + API 路由(start/stop/stats/download/export.csv) |
|
||||
| `collect.go` | 核心:签名、EAN-13 校验位、范围/清单枚举、并发+限速、JSONL 落库、断点续采 |
|
||||
| `web/index.html` | 内嵌(`go:embed`)的控制台界面 |
|
||||
| `build.sh` | 交叉编译出 `bypos-collector.exe`(windows/amd64)与 linux 测试二进制 |
|
||||
| `使用说明.md` | 面向使用者的操作说明 |
|
||||
|
||||
## 构建
|
||||
|
||||
```bash
|
||||
./build.sh
|
||||
# 产物:bypos-collector.exe(发给 Windows 用户)/ bypos-collector-linux(本地测试)
|
||||
```
|
||||
|
||||
二进制与采集产物(`*.jsonl`/`*.csv`)已在 `.gitignore` 中排除,不入库。
|
||||
|
||||
## 接口与签名(逆向所得,后期迭代参考)
|
||||
|
||||
云店新增商品页输入条码时,前端经服务端代理 `/prod-api/ZmSvr/httpUtil/getGet`
|
||||
转发到中心库:
|
||||
|
||||
```
|
||||
GET http://zc.bypos.net/byGoodsService/byMessage.asmx/GetGoodsinfo
|
||||
?sdogid=<账号id>®num=1&barcode=<条码>
|
||||
&sparm1=<md5(sdogid)> # 常量,随账号固定
|
||||
&sparm2=<md5(barcode + tsMs)> # tsMs = 当前秒*1000(末尾恒为 000)
|
||||
&sparm3=<tsMs 前 10 位 = 秒级时间戳>
|
||||
&sparm4=&barcodetype=yunpos
|
||||
```
|
||||
|
||||
返回 `<string>{...json...}</string>`,内层 JSON 字段:
|
||||
|
||||
| 上游字段 | 含义 | 归一化字段 |
|
||||
| --- | --- | --- |
|
||||
| item_name | 品名 | name |
|
||||
| item_size | 规格 | spec |
|
||||
| unit_no | 单位 | unit |
|
||||
| item_area | 产地/地区 | area |
|
||||
| birth_com | 生产企业(常空) | manufacturer |
|
||||
| birth_doc | 生产许可(常空) | license |
|
||||
| inprice | 建议进价 | in_price |
|
||||
| sellprice | 建议零售价 | sell_price |
|
||||
| retcode | 1=命中,0=失败 | status(hit/miss/invalid) |
|
||||
|
||||
`retmsg` 含「非国标条码 / 参数异常」=> invalid;含「条码不存在」=> miss。
|
||||
|
||||
## 配置
|
||||
|
||||
- `sdogid`:中心库账号 id(本项目所属云店账号的授权 id)。默认值见
|
||||
`collect.go` 的 `defaultSdogID`,也可用 `-sdogid` 参数或控制台覆盖。
|
||||
**这是账号级凭证**——若本仓库对外公开,建议改为从环境变量/外部配置读取。
|
||||
|
||||
## 注意
|
||||
|
||||
批量自动查询比页面逐条更"重",上游可能对账号限频。请低速、分前缀/品类分批采集。
|
||||
@@ -0,0 +1,11 @@
|
||||
#!/usr/bin/env bash
|
||||
# Build the bypos-collector for Windows (and a Linux binary for local testing).
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")"
|
||||
gofmt -w ./*.go
|
||||
go vet ./...
|
||||
echo "building windows/amd64 .exe ..."
|
||||
GOOS=windows GOARCH=amd64 go build -ldflags "-s -w" -o bypos-collector.exe .
|
||||
echo "building linux/amd64 (for testing) ..."
|
||||
go build -o bypos-collector-linux .
|
||||
ls -la bypos-collector.exe bypos-collector-linux
|
||||
@@ -0,0 +1,481 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/md5"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
)
|
||||
|
||||
// ---- upstream config ----
|
||||
// sdogid is the 云店 account license id observed in the live request. It is
|
||||
// configurable so the tool is not tied to a single account.
|
||||
const defaultSdogID = "137966"
|
||||
|
||||
const endpoint = "http://zc.bypos.net/byGoodsService/byMessage.asmx/GetGoodsinfo"
|
||||
|
||||
var stringTagRe = regexp.MustCompile(`(?s)<string[^>]*>(.*)</string>`)
|
||||
|
||||
// Product is the normalized record we persist (one JSON object per line).
|
||||
type Product struct {
|
||||
Barcode string `json:"barcode"`
|
||||
Name string `json:"name"` // item_name 品名
|
||||
Spec string `json:"spec"` // item_size 规格
|
||||
Unit string `json:"unit"` // unit_no 单位
|
||||
Area string `json:"area"` // item_area 产地/地区
|
||||
Manufacturer string `json:"manufacturer"` // birth_com 生产企业
|
||||
License string `json:"license"` // birth_doc 生产许可
|
||||
InPrice string `json:"in_price"` // 建议进价
|
||||
SellPrice string `json:"sell_price"` // 建议零售价
|
||||
Status string `json:"status"` // hit / miss / invalid / error
|
||||
RetMsg string `json:"retmsg"` // 原始返回信息
|
||||
FetchedAt string `json:"fetched_at"` // RFC3339
|
||||
Source string `json:"source"` // zc.bypos.net
|
||||
}
|
||||
|
||||
// upstream raw fields
|
||||
type rawResp struct {
|
||||
RetCode string `json:"retcode"`
|
||||
RetMsg string `json:"retmsg"`
|
||||
Barcode string `json:"barcode"`
|
||||
ItemName string `json:"item_name"`
|
||||
UnitNo string `json:"unit_no"`
|
||||
ItemSize string `json:"item_size"`
|
||||
ItemArea string `json:"item_area"`
|
||||
BirthCom string `json:"birth_com"`
|
||||
BirthDoc string `json:"birth_doc"`
|
||||
InPrice string `json:"inprice"`
|
||||
SellPrice string `json:"sellprice"`
|
||||
}
|
||||
|
||||
func md5hex(s string) string {
|
||||
h := md5.Sum([]byte(s))
|
||||
return hex.EncodeToString(h[:])
|
||||
}
|
||||
|
||||
// ean13Check computes the EAN-13 check digit for a 12-digit body.
|
||||
func ean13Check(body string) (string, bool) {
|
||||
if len(body) != 12 {
|
||||
return "", false
|
||||
}
|
||||
sum := 0
|
||||
for i := 0; i < 12; i++ {
|
||||
c := body[i]
|
||||
if c < '0' || c > '9' {
|
||||
return "", false
|
||||
}
|
||||
d := int(c - '0')
|
||||
if i%2 == 0 {
|
||||
sum += d
|
||||
} else {
|
||||
sum += d * 3
|
||||
}
|
||||
}
|
||||
chk := (10 - (sum % 10)) % 10
|
||||
return body + strconv.Itoa(chk), true
|
||||
}
|
||||
|
||||
// lookup queries the upstream central library for one barcode.
|
||||
func (c *Collector) lookup(ctx context.Context, barcode string) (*Product, error) {
|
||||
tsMs := strconv.FormatInt(time.Now().Unix()*1000, 10) // always ends in 000
|
||||
sparm1 := md5hex(c.sdogID)
|
||||
sparm2 := md5hex(barcode + tsMs)
|
||||
sparm3 := tsMs[:10]
|
||||
q := url.Values{}
|
||||
q.Set("sdogid", c.sdogID)
|
||||
q.Set("regnum", "1")
|
||||
q.Set("barcode", barcode)
|
||||
q.Set("sparm1", sparm1)
|
||||
q.Set("sparm2", sparm2)
|
||||
q.Set("sparm3", sparm3)
|
||||
q.Set("sparm4", "")
|
||||
q.Set("barcodetype", "yunpos")
|
||||
reqURL := endpoint + "?" + q.Encode()
|
||||
|
||||
req, _ := http.NewRequestWithContext(ctx, "GET", reqURL, nil)
|
||||
req.Header.Set("User-Agent", "Mozilla/5.0")
|
||||
resp, err := c.client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
b, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
inner := b
|
||||
if m := stringTagRe.FindSubmatch(b); m != nil {
|
||||
inner = m[1]
|
||||
}
|
||||
var r rawResp
|
||||
if err := json.Unmarshal(inner, &r); err != nil {
|
||||
return nil, fmt.Errorf("parse: %v (body=%.120s)", err, string(b))
|
||||
}
|
||||
p := &Product{
|
||||
Barcode: barcode,
|
||||
RetMsg: r.RetMsg,
|
||||
FetchedAt: time.Now().Format(time.RFC3339),
|
||||
Source: "zc.bypos.net",
|
||||
}
|
||||
if r.RetCode == "1" {
|
||||
p.Status = "hit"
|
||||
p.Name = strings.TrimSpace(r.ItemName)
|
||||
p.Spec = strings.TrimSpace(r.ItemSize)
|
||||
p.Unit = strings.TrimSpace(r.UnitNo)
|
||||
p.Area = strings.TrimSpace(r.ItemArea)
|
||||
p.Manufacturer = strings.TrimSpace(r.BirthCom)
|
||||
p.License = strings.TrimSpace(r.BirthDoc)
|
||||
p.InPrice = strings.TrimSpace(r.InPrice)
|
||||
p.SellPrice = strings.TrimSpace(r.SellPrice)
|
||||
} else if strings.Contains(r.RetMsg, "非国标") || strings.Contains(r.RetMsg, "参数异常") {
|
||||
p.Status = "invalid"
|
||||
} else {
|
||||
p.Status = "miss"
|
||||
}
|
||||
return p, nil
|
||||
}
|
||||
|
||||
// ---- job / collector state ----
|
||||
|
||||
type Stats struct {
|
||||
Running bool `json:"running"`
|
||||
Total int64 `json:"total"`
|
||||
Done int64 `json:"done"`
|
||||
Hits int64 `json:"hits"`
|
||||
Miss int64 `json:"miss"`
|
||||
Invalid int64 `json:"invalid"`
|
||||
Errors int64 `json:"errors"`
|
||||
Skipped int64 `json:"skipped"`
|
||||
Current string `json:"current"`
|
||||
OutFile string `json:"out_file"`
|
||||
StartedAt string `json:"started_at"`
|
||||
Message string `json:"message"`
|
||||
}
|
||||
|
||||
type Collector struct {
|
||||
mu sync.Mutex
|
||||
client *http.Client
|
||||
sdogID string
|
||||
outPath string
|
||||
outFile *os.File
|
||||
|
||||
cancel context.CancelFunc
|
||||
wg sync.WaitGroup
|
||||
|
||||
// atomic counters
|
||||
total, done, hits, miss, invalid, errors, skipped int64
|
||||
running int32
|
||||
|
||||
current atomic.Value // string
|
||||
startedAt string
|
||||
message string
|
||||
|
||||
seen map[string]struct{} // barcodes already in output (dedupe / resume)
|
||||
recent []Product // ring of last results for UI
|
||||
}
|
||||
|
||||
func NewCollector(sdogID string) *Collector {
|
||||
if sdogID == "" {
|
||||
sdogID = defaultSdogID
|
||||
}
|
||||
c := &Collector{
|
||||
client: &http.Client{Timeout: 25 * time.Second},
|
||||
sdogID: sdogID,
|
||||
seen: map[string]struct{}{},
|
||||
}
|
||||
c.current.Store("")
|
||||
return c
|
||||
}
|
||||
|
||||
func (c *Collector) isRunning() bool { return atomic.LoadInt32(&c.running) == 1 }
|
||||
|
||||
func (c *Collector) snapshot() Stats {
|
||||
cur, _ := c.current.Load().(string)
|
||||
c.mu.Lock()
|
||||
msg := c.message
|
||||
out := c.outPath
|
||||
started := c.startedAt
|
||||
c.mu.Unlock()
|
||||
return Stats{
|
||||
Running: c.isRunning(),
|
||||
Total: atomic.LoadInt64(&c.total),
|
||||
Done: atomic.LoadInt64(&c.done),
|
||||
Hits: atomic.LoadInt64(&c.hits),
|
||||
Miss: atomic.LoadInt64(&c.miss),
|
||||
Invalid: atomic.LoadInt64(&c.invalid),
|
||||
Errors: atomic.LoadInt64(&c.errors),
|
||||
Skipped: atomic.LoadInt64(&c.skipped),
|
||||
Current: cur,
|
||||
OutFile: out,
|
||||
StartedAt: started,
|
||||
Message: msg,
|
||||
}
|
||||
}
|
||||
|
||||
func (c *Collector) recentResults() []Product {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
out := make([]Product, len(c.recent))
|
||||
copy(out, c.recent)
|
||||
return out
|
||||
}
|
||||
|
||||
func (c *Collector) pushRecent(p Product) {
|
||||
c.mu.Lock()
|
||||
c.recent = append(c.recent, p)
|
||||
if len(c.recent) > 60 {
|
||||
c.recent = c.recent[len(c.recent)-60:]
|
||||
}
|
||||
c.mu.Unlock()
|
||||
}
|
||||
|
||||
// loadSeen reads an existing output file to build the dedupe set (for resume).
|
||||
func (c *Collector) loadSeen(path string) error {
|
||||
c.seen = map[string]struct{}{}
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
defer f.Close()
|
||||
dec := json.NewDecoder(f)
|
||||
for {
|
||||
var p Product
|
||||
if err := dec.Decode(&p); err != nil {
|
||||
break
|
||||
}
|
||||
if p.Barcode != "" {
|
||||
c.seen[p.Barcode] = struct{}{}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
type JobReq struct {
|
||||
Mode string `json:"mode"` // "range" | "list"
|
||||
StartBody string `json:"start_body"` // 12-digit body (range mode)
|
||||
EndBody string `json:"end_body"` // 12-digit body (range mode)
|
||||
List string `json:"list"` // newline/space separated barcodes (list mode)
|
||||
Concurrency int `json:"concurrency"` // parallel requests
|
||||
DelayMs int `json:"delay_ms"` // min interval between request starts
|
||||
LogMiss bool `json:"log_miss"` // also write miss/invalid lines
|
||||
OutFile string `json:"out_file"`
|
||||
SdogID string `json:"sdog_id"`
|
||||
}
|
||||
|
||||
func sanitizeBarcodes(s string) []string {
|
||||
fields := regexp.MustCompile(`[^0-9]+`).Split(s, -1)
|
||||
var out []string
|
||||
for _, f := range fields {
|
||||
if f != "" {
|
||||
out = append(out, f)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Start launches a collection job. Returns error if validation fails or busy.
|
||||
func (c *Collector) Start(req JobReq) error {
|
||||
if c.isRunning() {
|
||||
return fmt.Errorf("已有任务在运行")
|
||||
}
|
||||
if req.Concurrency <= 0 {
|
||||
req.Concurrency = 3
|
||||
}
|
||||
if req.Concurrency > 20 {
|
||||
req.Concurrency = 20
|
||||
}
|
||||
if req.DelayMs < 0 {
|
||||
req.DelayMs = 0
|
||||
}
|
||||
if req.OutFile == "" {
|
||||
req.OutFile = "products.jsonl"
|
||||
}
|
||||
if req.SdogID != "" {
|
||||
c.sdogID = req.SdogID
|
||||
}
|
||||
|
||||
// Build the list of barcodes to query.
|
||||
var barcodes []string
|
||||
switch req.Mode {
|
||||
case "list":
|
||||
barcodes = sanitizeBarcodes(req.List)
|
||||
if len(barcodes) == 0 {
|
||||
return fmt.Errorf("条码清单为空")
|
||||
}
|
||||
case "range":
|
||||
start, err := strconv.ParseInt(req.StartBody, 10, 64)
|
||||
if err != nil || len(req.StartBody) != 12 {
|
||||
return fmt.Errorf("起始码必须是 12 位数字(不含校验位)")
|
||||
}
|
||||
end, err := strconv.ParseInt(req.EndBody, 10, 64)
|
||||
if err != nil || len(req.EndBody) != 12 {
|
||||
return fmt.Errorf("结束码必须是 12 位数字(不含校验位)")
|
||||
}
|
||||
if end < start {
|
||||
return fmt.Errorf("结束码不能小于起始码")
|
||||
}
|
||||
if end-start+1 > 5_000_000 {
|
||||
return fmt.Errorf("单次范围过大(>500万),请缩小区间分批采集")
|
||||
}
|
||||
for v := start; v <= end; v++ {
|
||||
body := fmt.Sprintf("%012d", v)
|
||||
full, ok := ean13Check(body)
|
||||
if ok {
|
||||
barcodes = append(barcodes, full)
|
||||
}
|
||||
}
|
||||
default:
|
||||
return fmt.Errorf("未知模式: %s", req.Mode)
|
||||
}
|
||||
|
||||
abs, _ := filepath.Abs(req.OutFile)
|
||||
if err := c.loadSeen(abs); err != nil {
|
||||
return fmt.Errorf("读取已有文件失败: %v", err)
|
||||
}
|
||||
f, err := os.OpenFile(abs, os.O_CREATE|os.O_WRONLY|os.O_APPEND, 0644)
|
||||
if err != nil {
|
||||
return fmt.Errorf("打开输出文件失败: %v", err)
|
||||
}
|
||||
c.outFile = f
|
||||
c.outPath = abs
|
||||
|
||||
// reset counters
|
||||
atomic.StoreInt64(&c.total, int64(len(barcodes)))
|
||||
atomic.StoreInt64(&c.done, 0)
|
||||
atomic.StoreInt64(&c.hits, 0)
|
||||
atomic.StoreInt64(&c.miss, 0)
|
||||
atomic.StoreInt64(&c.invalid, 0)
|
||||
atomic.StoreInt64(&c.errors, 0)
|
||||
atomic.StoreInt64(&c.skipped, 0)
|
||||
c.mu.Lock()
|
||||
c.recent = nil
|
||||
c.startedAt = time.Now().Format(time.RFC3339)
|
||||
c.message = ""
|
||||
c.mu.Unlock()
|
||||
atomic.StoreInt32(&c.running, 1)
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
c.cancel = cancel
|
||||
|
||||
go c.run(ctx, barcodes, req)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *Collector) Stop() {
|
||||
if c.cancel != nil {
|
||||
c.cancel()
|
||||
}
|
||||
}
|
||||
|
||||
func (c *Collector) run(ctx context.Context, barcodes []string, req JobReq) {
|
||||
defer func() {
|
||||
atomic.StoreInt32(&c.running, 0)
|
||||
if c.outFile != nil {
|
||||
c.outFile.Sync()
|
||||
c.outFile.Close()
|
||||
c.outFile = nil
|
||||
}
|
||||
c.current.Store("")
|
||||
}()
|
||||
|
||||
jobs := make(chan string, req.Concurrency*2)
|
||||
var writeMu sync.Mutex
|
||||
|
||||
// global rate limiter: one token every DelayMs
|
||||
var ticker *time.Ticker
|
||||
if req.DelayMs > 0 {
|
||||
ticker = time.NewTicker(time.Duration(req.DelayMs) * time.Millisecond)
|
||||
defer ticker.Stop()
|
||||
}
|
||||
|
||||
worker := func() {
|
||||
defer c.wg.Done()
|
||||
for bc := range jobs {
|
||||
if ctx.Err() != nil {
|
||||
return
|
||||
}
|
||||
if ticker != nil {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
}
|
||||
c.current.Store(bc)
|
||||
p, err := c.lookup(ctx, bc)
|
||||
if err != nil {
|
||||
if ctx.Err() != nil {
|
||||
return
|
||||
}
|
||||
atomic.AddInt64(&c.errors, 1)
|
||||
atomic.AddInt64(&c.done, 1)
|
||||
ep := Product{Barcode: bc, Status: "error", RetMsg: err.Error(), FetchedAt: time.Now().Format(time.RFC3339), Source: "zc.bypos.net"}
|
||||
c.pushRecent(ep)
|
||||
continue
|
||||
}
|
||||
switch p.Status {
|
||||
case "hit":
|
||||
atomic.AddInt64(&c.hits, 1)
|
||||
case "miss":
|
||||
atomic.AddInt64(&c.miss, 1)
|
||||
case "invalid":
|
||||
atomic.AddInt64(&c.invalid, 1)
|
||||
}
|
||||
atomic.AddInt64(&c.done, 1)
|
||||
c.pushRecent(*p)
|
||||
if p.Status == "hit" || req.LogMiss {
|
||||
line, _ := json.Marshal(p)
|
||||
writeMu.Lock()
|
||||
c.outFile.Write(line)
|
||||
c.outFile.Write([]byte("\n"))
|
||||
c.seen[bc] = struct{}{}
|
||||
writeMu.Unlock()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for i := 0; i < req.Concurrency; i++ {
|
||||
c.wg.Add(1)
|
||||
go worker()
|
||||
}
|
||||
|
||||
for _, bc := range barcodes {
|
||||
if ctx.Err() != nil {
|
||||
break
|
||||
}
|
||||
if _, ok := c.seen[bc]; ok {
|
||||
atomic.AddInt64(&c.skipped, 1)
|
||||
atomic.AddInt64(&c.done, 1)
|
||||
continue
|
||||
}
|
||||
select {
|
||||
case jobs <- bc:
|
||||
case <-ctx.Done():
|
||||
}
|
||||
}
|
||||
close(jobs)
|
||||
c.wg.Wait()
|
||||
|
||||
c.mu.Lock()
|
||||
if ctx.Err() != nil {
|
||||
c.message = "已停止"
|
||||
} else {
|
||||
c.message = "采集完成"
|
||||
}
|
||||
c.mu.Unlock()
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
module byposcollector
|
||||
|
||||
go 1.23.4
|
||||
@@ -0,0 +1,156 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"embed"
|
||||
"encoding/csv"
|
||||
"encoding/json"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"log"
|
||||
"net"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"runtime"
|
||||
"time"
|
||||
)
|
||||
|
||||
//go:embed web/*
|
||||
var webFS embed.FS
|
||||
|
||||
var collector = NewCollector("")
|
||||
|
||||
func main() {
|
||||
addr := flag.String("addr", "127.0.0.1:8765", "本地监听地址")
|
||||
noOpen := flag.Bool("no-open", false, "不自动打开浏览器")
|
||||
sdog := flag.String("sdogid", "", "中心库账号 id(默认使用内置值)")
|
||||
flag.Parse()
|
||||
if *sdog != "" {
|
||||
collector.sdogID = *sdog
|
||||
}
|
||||
|
||||
sub, _ := fs.Sub(webFS, "web")
|
||||
mux := http.NewServeMux()
|
||||
mux.Handle("/", http.FileServer(http.FS(sub)))
|
||||
mux.HandleFunc("/api/start", handleStart)
|
||||
mux.HandleFunc("/api/stop", handleStop)
|
||||
mux.HandleFunc("/api/stats", handleStats)
|
||||
mux.HandleFunc("/api/download", handleDownload)
|
||||
mux.HandleFunc("/api/export.csv", handleExportCSV)
|
||||
|
||||
ln, err := net.Listen("tcp", *addr)
|
||||
if err != nil {
|
||||
log.Fatalf("无法监听 %s: %v", *addr, err)
|
||||
}
|
||||
realAddr := ln.Addr().String()
|
||||
urlStr := "http://" + realAddr + "/"
|
||||
fmt.Println("==============================================")
|
||||
fmt.Println(" 中心库商品采集器 bypos-collector")
|
||||
fmt.Println(" 控制台: " + urlStr)
|
||||
fmt.Println(" 关闭本窗口即停止程序")
|
||||
fmt.Println("==============================================")
|
||||
if !*noOpen {
|
||||
go openBrowser(urlStr)
|
||||
}
|
||||
log.Fatal(http.Serve(ln, mux))
|
||||
}
|
||||
|
||||
func openBrowser(url string) {
|
||||
time.Sleep(600 * time.Millisecond)
|
||||
var cmd string
|
||||
var args []string
|
||||
switch runtime.GOOS {
|
||||
case "windows":
|
||||
cmd = "rundll32"
|
||||
args = []string{"url.dll,FileProtocolHandler", url}
|
||||
case "darwin":
|
||||
cmd = "open"
|
||||
args = []string{url}
|
||||
default:
|
||||
cmd = "xdg-open"
|
||||
args = []string{url}
|
||||
}
|
||||
_ = exec.Command(cmd, args...).Start()
|
||||
}
|
||||
|
||||
func writeJSON(w http.ResponseWriter, code int, v interface{}) {
|
||||
w.Header().Set("Content-Type", "application/json; charset=utf-8")
|
||||
w.WriteHeader(code)
|
||||
json.NewEncoder(w).Encode(v)
|
||||
}
|
||||
|
||||
func handleStart(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != "POST" {
|
||||
http.Error(w, "method", 405)
|
||||
return
|
||||
}
|
||||
var req JobReq
|
||||
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
|
||||
writeJSON(w, 400, map[string]string{"error": "请求格式错误"})
|
||||
return
|
||||
}
|
||||
if err := collector.Start(req); err != nil {
|
||||
writeJSON(w, 400, map[string]string{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
writeJSON(w, 200, map[string]string{"ok": "started"})
|
||||
}
|
||||
|
||||
func handleStop(w http.ResponseWriter, r *http.Request) {
|
||||
collector.Stop()
|
||||
writeJSON(w, 200, map[string]string{"ok": "stopping"})
|
||||
}
|
||||
|
||||
func handleStats(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, 200, map[string]interface{}{
|
||||
"stats": collector.snapshot(),
|
||||
"recent": collector.recentResults(),
|
||||
})
|
||||
}
|
||||
|
||||
func handleDownload(w http.ResponseWriter, r *http.Request) {
|
||||
s := collector.snapshot()
|
||||
if s.OutFile == "" {
|
||||
http.Error(w, "no output yet", 404)
|
||||
return
|
||||
}
|
||||
f, err := os.Open(s.OutFile)
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), 404)
|
||||
return
|
||||
}
|
||||
defer f.Close()
|
||||
w.Header().Set("Content-Type", "application/x-ndjson; charset=utf-8")
|
||||
w.Header().Set("Content-Disposition", "attachment; filename=products.jsonl")
|
||||
io.Copy(w, f)
|
||||
}
|
||||
|
||||
func handleExportCSV(w http.ResponseWriter, r *http.Request) {
|
||||
s := collector.snapshot()
|
||||
if s.OutFile == "" {
|
||||
http.Error(w, "no output yet", 404)
|
||||
return
|
||||
}
|
||||
f, err := os.Open(s.OutFile)
|
||||
if err != nil {
|
||||
http.Error(w, err.Error(), 404)
|
||||
return
|
||||
}
|
||||
defer f.Close()
|
||||
w.Header().Set("Content-Type", "text/csv; charset=utf-8")
|
||||
w.Header().Set("Content-Disposition", "attachment; filename=products.csv")
|
||||
w.Write([]byte{0xEF, 0xBB, 0xBF}) // UTF-8 BOM so Excel reads Chinese correctly
|
||||
cw := csv.NewWriter(w)
|
||||
cw.Write([]string{"barcode", "name", "spec", "unit", "area", "manufacturer", "license", "in_price", "sell_price", "status", "fetched_at"})
|
||||
dec := json.NewDecoder(f)
|
||||
for {
|
||||
var p Product
|
||||
if err := dec.Decode(&p); err != nil {
|
||||
break
|
||||
}
|
||||
cw.Write([]string{p.Barcode, p.Name, p.Spec, p.Unit, p.Area, p.Manufacturer, p.License, p.InPrice, p.SellPrice, p.Status, p.FetchedAt})
|
||||
}
|
||||
cw.Flush()
|
||||
}
|
||||
@@ -0,0 +1,206 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="zh-CN">
|
||||
<head>
|
||||
<meta charset="utf-8"/>
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1"/>
|
||||
<title>中心库商品采集器</title>
|
||||
<style>
|
||||
* { box-sizing: border-box; }
|
||||
body { font-family: -apple-system, "Microsoft YaHei", Arial, sans-serif; margin: 0; background:#f4f6f9; color:#222; }
|
||||
header { background:#1f6feb; color:#fff; padding:14px 22px; font-size:18px; font-weight:600; }
|
||||
.wrap { max-width:1080px; margin:18px auto; padding:0 16px; }
|
||||
.card { background:#fff; border:1px solid #e3e8ef; border-radius:10px; padding:18px 20px; margin-bottom:16px; }
|
||||
.card h3 { margin:0 0 12px; font-size:15px; color:#1f6feb; }
|
||||
label { display:block; font-size:13px; color:#555; margin:8px 0 4px; }
|
||||
input[type=text], input[type=number], textarea, select {
|
||||
width:100%; padding:8px 10px; border:1px solid #cdd5e0; border-radius:6px; font-size:14px;
|
||||
}
|
||||
textarea { height:90px; font-family:monospace; }
|
||||
.row { display:flex; gap:14px; flex-wrap:wrap; }
|
||||
.row > div { flex:1; min-width:160px; }
|
||||
.tabs { display:flex; gap:8px; margin-bottom:12px; }
|
||||
.tab { padding:7px 16px; border:1px solid #cdd5e0; border-radius:20px; cursor:pointer; font-size:13px; background:#fff; }
|
||||
.tab.active { background:#1f6feb; color:#fff; border-color:#1f6feb; }
|
||||
button.primary { background:#1f6feb; color:#fff; border:none; padding:10px 22px; border-radius:6px; font-size:14px; cursor:pointer; }
|
||||
button.danger { background:#d1242f; color:#fff; border:none; padding:10px 22px; border-radius:6px; font-size:14px; cursor:pointer; }
|
||||
button.ghost { background:#fff; color:#1f6feb; border:1px solid #1f6feb; padding:8px 16px; border-radius:6px; cursor:pointer; font-size:13px; }
|
||||
button:disabled { opacity:.5; cursor:not-allowed; }
|
||||
.stats { display:flex; gap:10px; flex-wrap:wrap; }
|
||||
.stat { flex:1; min-width:90px; background:#f7f9fc; border:1px solid #e3e8ef; border-radius:8px; padding:10px; text-align:center; }
|
||||
.stat .n { font-size:22px; font-weight:700; }
|
||||
.stat .l { font-size:12px; color:#777; margin-top:2px; }
|
||||
.bar { height:10px; background:#e3e8ef; border-radius:6px; overflow:hidden; margin:10px 0; }
|
||||
.bar > div { height:100%; background:#2da44e; width:0%; transition:width .4s; }
|
||||
table { width:100%; border-collapse:collapse; font-size:13px; }
|
||||
th, td { text-align:left; padding:6px 8px; border-bottom:1px solid #eef1f5; white-space:nowrap; overflow:hidden; text-overflow:ellipsis; max-width:180px; }
|
||||
th { color:#888; font-weight:600; }
|
||||
.hit { color:#2da44e; } .miss { color:#999; } .invalid { color:#d1242f; } .error { color:#bf8700; }
|
||||
.hint { font-size:12px; color:#888; margin-top:6px; line-height:1.5; }
|
||||
.est { font-size:13px; color:#1f6feb; margin-top:6px; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<header>中心库商品采集器 · bypos-collector</header>
|
||||
<div class="wrap">
|
||||
|
||||
<div class="card">
|
||||
<h3>① 规划采集范围</h3>
|
||||
<div class="tabs">
|
||||
<div class="tab active" data-mode="range" onclick="setMode('range')">按条码范围</div>
|
||||
<div class="tab" data-mode="list" onclick="setMode('list')">按条码清单</div>
|
||||
</div>
|
||||
|
||||
<div id="pane-range">
|
||||
<div class="hint">EAN-13 国标条码共 13 位,最后一位是校验位由程序自动计算。下面填<b>前 12 位</b>(本体),程序逐个枚举并补校验位查询。常见前缀:69 开头为中国大陆。</div>
|
||||
<label>快捷填充前缀(可选)</label>
|
||||
<div class="row">
|
||||
<div><input type="text" id="prefix" placeholder="如 690100,点下方按钮自动算区间"/></div>
|
||||
<div style="flex:0"><button class="ghost" onclick="fillFromPrefix()">用前缀填充区间</button></div>
|
||||
</div>
|
||||
<div class="row">
|
||||
<div>
|
||||
<label>起始本体(12 位)</label>
|
||||
<input type="text" id="start" value="690100000000" maxlength="12"/>
|
||||
</div>
|
||||
<div>
|
||||
<label>结束本体(12 位)</label>
|
||||
<input type="text" id="end" value="690100000999" maxlength="12"/>
|
||||
</div>
|
||||
</div>
|
||||
<div class="est" id="est"></div>
|
||||
</div>
|
||||
|
||||
<div id="pane-list" style="display:none">
|
||||
<label>粘贴条码清单(每行一个,或用空格/逗号分隔)</label>
|
||||
<textarea id="list" placeholder="6901028941068 6920202888883"></textarea>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h3>② 采集参数</h3>
|
||||
<div class="row">
|
||||
<div>
|
||||
<label>并发数</label>
|
||||
<input type="number" id="concurrency" value="3" min="1" max="20"/>
|
||||
</div>
|
||||
<div>
|
||||
<label>每次请求间隔(毫秒)</label>
|
||||
<input type="number" id="delay" value="300" min="0"/>
|
||||
</div>
|
||||
<div>
|
||||
<label>输出文件名</label>
|
||||
<input type="text" id="outfile" value="products.jsonl"/>
|
||||
</div>
|
||||
</div>
|
||||
<label style="margin-top:12px"><input type="checkbox" id="logmiss"/> 同时记录未命中/无效条码(默认只存命中)</label>
|
||||
<div class="hint">速度越快越容易触发上游频控。建议并发 3、间隔 300ms 起步,稳定后再调。已采过的条码会自动跳过(断点续采)。</div>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h3>③ 运行</h3>
|
||||
<div style="margin-bottom:12px">
|
||||
<button class="primary" id="btnStart" onclick="start()">开始采集</button>
|
||||
<button class="danger" id="btnStop" onclick="stop()" disabled>停止</button>
|
||||
<button class="ghost" onclick="window.open('/api/download')">下载 JSONL</button>
|
||||
<button class="ghost" onclick="window.open('/api/export.csv')">导出 CSV(Excel)</button>
|
||||
</div>
|
||||
<div class="bar"><div id="prog"></div></div>
|
||||
<div class="stats">
|
||||
<div class="stat"><div class="n" id="s-done">0</div><div class="l">已处理</div></div>
|
||||
<div class="stat"><div class="n" id="s-total">0</div><div class="l">总计</div></div>
|
||||
<div class="stat"><div class="n hit" id="s-hits">0</div><div class="l">命中</div></div>
|
||||
<div class="stat"><div class="n miss" id="s-miss">0</div><div class="l">未命中</div></div>
|
||||
<div class="stat"><div class="n invalid" id="s-invalid">0</div><div class="l">无效</div></div>
|
||||
<div class="stat"><div class="n error" id="s-errors">0</div><div class="l">错误</div></div>
|
||||
<div class="stat"><div class="n" id="s-skipped">0</div><div class="l">跳过</div></div>
|
||||
</div>
|
||||
<div class="hint" id="msg"></div>
|
||||
</div>
|
||||
|
||||
<div class="card">
|
||||
<h3>④ 实时结果(最近 60 条)</h3>
|
||||
<div style="max-height:340px; overflow:auto">
|
||||
<table>
|
||||
<thead><tr><th>条码</th><th>品名</th><th>规格</th><th>单位</th><th>产地</th><th>进价</th><th>零售价</th><th>状态</th></tr></thead>
|
||||
<tbody id="rows"></tbody>
|
||||
</table>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
let mode = 'range';
|
||||
function setMode(m){
|
||||
mode = m;
|
||||
document.querySelectorAll('.tab').forEach(t=>t.classList.toggle('active', t.dataset.mode===m));
|
||||
document.getElementById('pane-range').style.display = m==='range'?'block':'none';
|
||||
document.getElementById('pane-list').style.display = m==='list'?'block':'none';
|
||||
}
|
||||
function fillFromPrefix(){
|
||||
let p = document.getElementById('prefix').value.replace(/[^0-9]/g,'');
|
||||
if(!p){ alert('请先填前缀'); return; }
|
||||
if(p.length>=12){ alert('前缀太长,应少于 12 位'); return; }
|
||||
let pad = 12 - p.length;
|
||||
document.getElementById('start').value = p + '0'.repeat(pad);
|
||||
document.getElementById('end').value = p + '9'.repeat(pad);
|
||||
updateEst();
|
||||
}
|
||||
function updateEst(){
|
||||
let s = document.getElementById('start').value.replace(/[^0-9]/g,'');
|
||||
let e = document.getElementById('end').value.replace(/[^0-9]/g,'');
|
||||
if(s.length===12 && e.length===12){
|
||||
let n = (BigInt(e) - BigInt(s)) + 1n;
|
||||
document.getElementById('est').textContent = '本次将查询约 ' + n.toString() + ' 个条码';
|
||||
} else {
|
||||
document.getElementById('est').textContent = '';
|
||||
}
|
||||
}
|
||||
document.getElementById('start').addEventListener('input', updateEst);
|
||||
document.getElementById('end').addEventListener('input', updateEst);
|
||||
updateEst();
|
||||
|
||||
async function start(){
|
||||
let body = {
|
||||
mode: mode,
|
||||
start_body: document.getElementById('start').value.trim(),
|
||||
end_body: document.getElementById('end').value.trim(),
|
||||
list: document.getElementById('list').value,
|
||||
concurrency: parseInt(document.getElementById('concurrency').value)||3,
|
||||
delay_ms: parseInt(document.getElementById('delay').value)||0,
|
||||
log_miss: document.getElementById('logmiss').checked,
|
||||
out_file: document.getElementById('outfile').value.trim()
|
||||
};
|
||||
let r = await fetch('/api/start', {method:'POST', headers:{'Content-Type':'application/json'}, body:JSON.stringify(body)});
|
||||
let j = await r.json();
|
||||
if(j.error){ alert('启动失败: ' + j.error); return; }
|
||||
}
|
||||
async function stop(){ await fetch('/api/stop', {method:'POST'}); }
|
||||
|
||||
function esc(s){ return (s||'').replace(/[&<>]/g, c=>({'&':'&','<':'<','>':'>'}[c])); }
|
||||
|
||||
async function poll(){
|
||||
try{
|
||||
let r = await fetch('/api/stats'); let j = await r.json();
|
||||
let s = j.stats;
|
||||
document.getElementById('s-done').textContent = s.done;
|
||||
document.getElementById('s-total').textContent = s.total;
|
||||
document.getElementById('s-hits').textContent = s.hits;
|
||||
document.getElementById('s-miss').textContent = s.miss;
|
||||
document.getElementById('s-invalid').textContent = s.invalid;
|
||||
document.getElementById('s-errors').textContent = s.errors;
|
||||
document.getElementById('s-skipped').textContent = s.skipped;
|
||||
let pct = s.total>0 ? Math.floor(s.done*100/s.total) : 0;
|
||||
document.getElementById('prog').style.width = pct + '%';
|
||||
document.getElementById('msg').textContent = (s.running? ('采集中… 当前 '+s.current) : (s.message||'空闲'));
|
||||
document.getElementById('btnStart').disabled = s.running;
|
||||
document.getElementById('btnStop').disabled = !s.running;
|
||||
let rows = (j.recent||[]).slice().reverse().map(p=>
|
||||
'<tr><td>'+esc(p.barcode)+'</td><td>'+esc(p.name)+'</td><td>'+esc(p.spec)+'</td><td>'+esc(p.unit)+'</td><td>'+esc(p.area)+'</td><td>'+esc(p.in_price)+'</td><td>'+esc(p.sell_price)+'</td><td class="'+p.status+'">'+esc(p.status)+'</td></tr>'
|
||||
).join('');
|
||||
document.getElementById('rows').innerHTML = rows;
|
||||
}catch(e){}
|
||||
}
|
||||
setInterval(poll, 1000); poll();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,68 @@
|
||||
# 中心库商品采集器 bypos-collector 使用说明
|
||||
|
||||
一个单文件 Windows 小程序,通过云店「新增商品」用到的同一个中心商品库
|
||||
(`zc.bypos.net`)按条码批量采集商品档案(品名/规格/单位/产地/厂商/建议进价/建议零售价),
|
||||
存到本地,供后续导入天工(goods)系统。
|
||||
|
||||
## 一、运行
|
||||
|
||||
1. 把 `bypos-collector.exe` 放到任意空文件夹(采集结果会生成在同一文件夹)。
|
||||
2. 双击运行。会弹出一个黑色命令行窗口(不要关它),并自动打开浏览器控制台
|
||||
`http://127.0.0.1:8765/`。
|
||||
- 若没自动打开,手动在浏览器输入上面这个地址。
|
||||
3. 用完直接关掉那个命令行窗口即可退出。
|
||||
|
||||
## 二、采集
|
||||
|
||||
控制台分四步:
|
||||
|
||||
**① 规划采集范围** —— 两种方式二选一:
|
||||
- **按条码范围**:EAN-13 国标条码共 13 位,最后一位是校验位,程序自动算。
|
||||
你只填**前 12 位**的起止区间。可在「快捷填充前缀」里填如 `690100`,
|
||||
点按钮自动生成区间(`690100000000` ~ `690100999999`)。
|
||||
- **按条码清单**:直接粘贴一批条码(每行一个,或空格/逗号分隔)。
|
||||
|
||||
**② 采集参数**:
|
||||
- 并发数(默认 3)、请求间隔(默认 300ms):**越慢越安全**,上游可能对账号限频。
|
||||
- 输出文件名(默认 `products.jsonl`)。
|
||||
- 「同时记录未命中/无效条码」:默认只存命中的;勾上会把未命中也记下来。
|
||||
|
||||
**③ 运行**:点「开始采集」。进度、命中/未命中/错误实时显示。
|
||||
已经采过的条码会自动跳过(可随时停了再开,断点续采)。
|
||||
|
||||
**④ 实时结果**:最近 60 条滚动显示。
|
||||
|
||||
## 三、导出
|
||||
|
||||
- 「下载 JSONL」:原始数据(每行一个 JSON),用于导入天工系统。
|
||||
- 「导出 CSV」:Excel 可直接打开查看。
|
||||
|
||||
## 四、字段说明(JSONL 每行)
|
||||
|
||||
| 字段 | 含义 |
|
||||
| --- | --- |
|
||||
| barcode | 条码(GTIN/EAN-13) |
|
||||
| name | 品名 |
|
||||
| spec | 规格 |
|
||||
| unit | 单位 |
|
||||
| area | 产地/地区 |
|
||||
| manufacturer | 生产企业(常为空) |
|
||||
| license | 生产许可(常为空) |
|
||||
| in_price | 建议进价 |
|
||||
| sell_price | 建议零售价 |
|
||||
| status | hit=命中 / miss=不存在 / invalid=非国标条码 / error=请求出错 |
|
||||
| fetched_at | 采集时间 |
|
||||
|
||||
## 五、注意
|
||||
|
||||
- 这是用云店账号授权去查上游中心库,**批量自动**比页面里一条条查更"重",
|
||||
上游厂商可能对账号做频控/限额。请低速、分批("一点点采"),发现大量报错就降速。
|
||||
- 全量 69 段是个天文数字,不要无脑全跑;建议按你关心的品牌/品类前缀分批。
|
||||
|
||||
## 六、命令行参数(可选)
|
||||
|
||||
```
|
||||
bypos-collector.exe -addr 127.0.0.1:8765 # 改监听端口
|
||||
bypos-collector.exe -no-open # 不自动开浏览器
|
||||
bypos-collector.exe -sdogid 137966 # 指定中心库账号 id
|
||||
```
|
||||
Reference in New Issue
Block a user